{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:UWDYL5OXYIJGNUTAHAROKRB5II","short_pith_number":"pith:UWDYL5OX","schema_version":"1.0","canonical_sha256":"a58785f5d7c21266d2603822e5443d42350d156f633a532e36b06672b6ace19f","source":{"kind":"arxiv","id":"2410.14072","version":1},"attestation_state":"computed","paper":{"title":"Efficient Vision-Language Models by Summarizing Visual Tokens into Compact Registers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Mahyar Najibi, Qichen Fu, Qingqing Cao, Sachin Mehta, Yuxin Wen","submitted_at":"2024-10-17T22:45:13Z","abstract_excerpt":"Recent advancements in vision-language models (VLMs) have expanded their potential for real-world applications, enabling these models to perform complex reasoning on images. In the widely used fully autoregressive transformer-based models like LLaVA, projected visual tokens are prepended to textual tokens. Oftentimes, visual tokens are significantly more than prompt tokens, resulting in increased computational overhead during both training and inference. In this paper, we propose Visual Compact Token Registers (Victor), a method that reduces the number of visual tokens by summarizing them into"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.14072","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-10-17T22:45:13Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"3c834d624c6a73ed5f6e5741defe9924d09c65f37945d6e9b2246e11bc6e9645","abstract_canon_sha256":"3759a375a50aa14c807db0dd24acfa4db7c0e8f23a80251f8848d1290627a8ba"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:22:26.710624Z","signature_b64":"1zbnBIOzE5YiAuAzwJAFGMtYSuK7fO277SaxxMud7K5J9yMfVqytq3I5St9ltDZQMVGEMlAInfJY817iXFA1BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a58785f5d7c21266d2603822e5443d42350d156f633a532e36b06672b6ace19f","last_reissued_at":"2026-07-05T09:22:26.710049Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:22:26.710049Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficient Vision-Language Models by Summarizing Visual Tokens into Compact Registers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Mahyar Najibi, Qichen Fu, Qingqing Cao, Sachin Mehta, Yuxin Wen","submitted_at":"2024-10-17T22:45:13Z","abstract_excerpt":"Recent advancements in vision-language models (VLMs) have expanded their potential for real-world applications, enabling these models to perform complex reasoning on images. In the widely used fully autoregressive transformer-based models like LLaVA, projected visual tokens are prepended to textual tokens. Oftentimes, visual tokens are significantly more than prompt tokens, resulting in increased computational overhead during both training and inference. In this paper, we propose Visual Compact Token Registers (Victor), a method that reduces the number of visual tokens by summarizing them into"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.14072","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.14072/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.14072","created_at":"2026-07-05T09:22:26.710106+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.14072v1","created_at":"2026-07-05T09:22:26.710106+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.14072","created_at":"2026-07-05T09:22:26.710106+00:00"},{"alias_kind":"pith_short_12","alias_value":"UWDYL5OXYIJG","created_at":"2026-07-05T09:22:26.710106+00:00"},{"alias_kind":"pith_short_16","alias_value":"UWDYL5OXYIJGNUTA","created_at":"2026-07-05T09:22:26.710106+00:00"},{"alias_kind":"pith_short_8","alias_value":"UWDYL5OX","created_at":"2026-07-05T09:22:26.710106+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05859","citing_title":"AVA-VLM: Adaptive Visual Attention-Vision Language Model for In-the-Wild Construction Site Monitoring","ref_index":40,"is_internal_anchor":true},{"citing_arxiv_id":"2605.16638","citing_title":"TTE-Flash: Accelerating Reasoning-based Multimodal Representations via Think-Then-Embed Tokens","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2603.26041","citing_title":"Where and How to Prune: An Empirical Study of Visual Token Pruning for GUI Agent Navigation","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2601.03233","citing_title":"LTX-2: Efficient Joint Audio-Visual Foundation Model","ref_index":30,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UWDYL5OXYIJGNUTAHAROKRB5II","json":"https://pith.science/pith/UWDYL5OXYIJGNUTAHAROKRB5II.json","graph_json":"https://pith.science/api/pith-number/UWDYL5OXYIJGNUTAHAROKRB5II/graph.json","events_json":"https://pith.science/api/pith-number/UWDYL5OXYIJGNUTAHAROKRB5II/events.json","paper":"https://pith.science/paper/UWDYL5OX"},"agent_actions":{"view_html":"https://pith.science/pith/UWDYL5OXYIJGNUTAHAROKRB5II","download_json":"https://pith.science/pith/UWDYL5OXYIJGNUTAHAROKRB5II.json","view_paper":"https://pith.science/paper/UWDYL5OX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.14072&json=true","fetch_graph":"https://pith.science/api/pith-number/UWDYL5OXYIJGNUTAHAROKRB5II/graph.json","fetch_events":"https://pith.science/api/pith-number/UWDYL5OXYIJGNUTAHAROKRB5II/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UWDYL5OXYIJGNUTAHAROKRB5II/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UWDYL5OXYIJGNUTAHAROKRB5II/action/storage_attestation","attest_author":"https://pith.science/pith/UWDYL5OXYIJGNUTAHAROKRB5II/action/author_attestation","sign_citation":"https://pith.science/pith/UWDYL5OXYIJGNUTAHAROKRB5II/action/citation_signature","submit_replication":"https://pith.science/pith/UWDYL5OXYIJGNUTAHAROKRB5II/action/replication_record"}},"created_at":"2026-07-05T09:22:26.710106+00:00","updated_at":"2026-07-05T09:22:26.710106+00:00"}