{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:B3O45SHHFEPCQESVICYTUSZAKM","short_pith_number":"pith:B3O45SHH","schema_version":"1.0","canonical_sha256":"0eddcec8e7291e28125540b13a4b20530ce4951e287a90082f70d328012cb718","source":{"kind":"arxiv","id":"2501.03895","version":2},"attestation_state":"computed","paper":{"title":"LLaVA-Mini: Efficient Image and Video Large Multimodal Models with One Vision Token","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Qingkai Fang, Shaolei Zhang, Yang Feng, Zhe Yang","submitted_at":"2025-01-07T16:03:14Z","abstract_excerpt":"The advent of real-time large multimodal models (LMMs) like GPT-4o has sparked considerable interest in efficient LMMs. LMM frameworks typically encode visual inputs into vision tokens (continuous representations) and integrate them and textual instructions into the context of large language models (LLMs), where large-scale parameters and numerous context tokens (predominantly vision tokens) result in substantial computational overhead. Previous efforts towards efficient LMMs always focus on replacing the LLM backbone with smaller models, while neglecting the crucial issue of token quantity. I"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.03895","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-01-07T16:03:14Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"eeae5623defa9558469697e02bc356b32586b91ca0199f536ff0326104c3e498","abstract_canon_sha256":"59e5b28c08699475212126ff4b19f5eae025e8073bc1f980519ac6853339c390"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:22:11.716224Z","signature_b64":"qNPChSFwhTAOiVDWIuAlJnWbkRxFcFZl54f8f9lYt7ZmoWSLBUYBYD4RZ8fmAga1z4vA89dUFTQjQk3zxwjWAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0eddcec8e7291e28125540b13a4b20530ce4951e287a90082f70d328012cb718","last_reissued_at":"2026-07-05T10:22:11.715690Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:22:11.715690Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLaVA-Mini: Efficient Image and Video Large Multimodal Models with One Vision Token","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Qingkai Fang, Shaolei Zhang, Yang Feng, Zhe Yang","submitted_at":"2025-01-07T16:03:14Z","abstract_excerpt":"The advent of real-time large multimodal models (LMMs) like GPT-4o has sparked considerable interest in efficient LMMs. LMM frameworks typically encode visual inputs into vision tokens (continuous representations) and integrate them and textual instructions into the context of large language models (LLMs), where large-scale parameters and numerous context tokens (predominantly vision tokens) result in substantial computational overhead. Previous efforts towards efficient LMMs always focus on replacing the LLM backbone with smaller models, while neglecting the crucial issue of token quantity. I"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.03895","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.03895/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.03895","created_at":"2026-07-05T10:22:11.715750+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.03895v2","created_at":"2026-07-05T10:22:11.715750+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.03895","created_at":"2026-07-05T10:22:11.715750+00:00"},{"alias_kind":"pith_short_12","alias_value":"B3O45SHHFEPC","created_at":"2026-07-05T10:22:11.715750+00:00"},{"alias_kind":"pith_short_16","alias_value":"B3O45SHHFEPCQESV","created_at":"2026-07-05T10:22:11.715750+00:00"},{"alias_kind":"pith_short_8","alias_value":"B3O45SHH","created_at":"2026-07-05T10:22:11.715750+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06468","citing_title":"EgoPolice: A Benchmark for Egocentric Video Understanding in High-Stakes Police Body-Worn Camera Footage","ref_index":83,"is_internal_anchor":true},{"citing_arxiv_id":"2606.20280","citing_title":"ELVA: Exploring Ranking-Driven Universal Multimodal Retrieval","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20077","citing_title":"The Hidden Evolution of Disguised Visual Context inside the VLM","ref_index":81,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31383","citing_title":"MS-Resampler: Multi-Scope Visual Resampling for Efficient Multimodal LLMs","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25952","citing_title":"VEN-VL: A Visual Ensemble MoE Framework for Effective and Efficient Multi-Modal Understanding","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28115","citing_title":"CIVIC: End-to-End Sequence Compactness for Efficient Vision-Language Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2508.06038","citing_title":"Fourier Compressor: Frequency-Domain Visual Token Compression for Vision-Language Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11530","citing_title":"Beyond Attention Scores: SVD-Based Vision Token Pruning for Efficient Vision-Language Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20950","citing_title":"Focus-then-Context: Subject-Centric Progressive Visual Token Reduction for Vision-Language Models","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17283","citing_title":"OProver: A Unified Framework for Agentic Formal Theorem Proving","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2509.09794","citing_title":"Synthetic Homes: A Multimodal Generative AI Pipeline for Residential Building Data Generation under Data Scarcity","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05899","citing_title":"VisMMOE: Exploiting Visual-Expert Affinity for Efficient Visual-Language MoE Offloading","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11627","citing_title":"POINTS-Long: Adaptive Dual-Mode Visual Reasoning in MLLMs","ref_index":115,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11530","citing_title":"Beyond Attention Scores: SVD-Based Vision Token Pruning for Efficient Vision-Language Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06809","citing_title":"LookWhen? Fast Video Recognition by Learning When, Where, and What to Compute","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18260","citing_title":"Geometry-Guided 3D Visual Token Pruning for Video-Language Models","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B3O45SHHFEPCQESVICYTUSZAKM","json":"https://pith.science/pith/B3O45SHHFEPCQESVICYTUSZAKM.json","graph_json":"https://pith.science/api/pith-number/B3O45SHHFEPCQESVICYTUSZAKM/graph.json","events_json":"https://pith.science/api/pith-number/B3O45SHHFEPCQESVICYTUSZAKM/events.json","paper":"https://pith.science/paper/B3O45SHH"},"agent_actions":{"view_html":"https://pith.science/pith/B3O45SHHFEPCQESVICYTUSZAKM","download_json":"https://pith.science/pith/B3O45SHHFEPCQESVICYTUSZAKM.json","view_paper":"https://pith.science/paper/B3O45SHH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.03895&json=true","fetch_graph":"https://pith.science/api/pith-number/B3O45SHHFEPCQESVICYTUSZAKM/graph.json","fetch_events":"https://pith.science/api/pith-number/B3O45SHHFEPCQESVICYTUSZAKM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B3O45SHHFEPCQESVICYTUSZAKM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B3O45SHHFEPCQESVICYTUSZAKM/action/storage_attestation","attest_author":"https://pith.science/pith/B3O45SHHFEPCQESVICYTUSZAKM/action/author_attestation","sign_citation":"https://pith.science/pith/B3O45SHHFEPCQESVICYTUSZAKM/action/citation_signature","submit_replication":"https://pith.science/pith/B3O45SHHFEPCQESVICYTUSZAKM/action/replication_record"}},"created_at":"2026-07-05T10:22:11.715750+00:00","updated_at":"2026-07-05T10:22:11.715750+00:00"}