{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:4AEZOLYJYBNDWFWEXTRKT3QHB3","short_pith_number":"pith:4AEZOLYJ","schema_version":"1.0","canonical_sha256":"e009972f09c05a3b16c4bce2a9ee070ecf4466f043e7d4b4e0a64f6b95d0255f","source":{"kind":"arxiv","id":"2606.12125","version":1},"attestation_state":"computed","paper":{"title":"Q-Fold: Query-Aware Focus-Context Spatio-Temporal Folding for Long Video Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Biao Tang, Chenqiang Gao, Jingyi Yuan, Shuxiang Gou, Xu Chen, Yuhan Zhang","submitted_at":"2026-06-10T14:19:15Z","abstract_excerpt":"Long-video understanding remains challenging for multimodal large language models, because temporally extended videos often contain thousands of frames and are therefore expensive to process exhaustively. Existing methods usually construct compact visual inputs from long videos under a limited visual budget. However, most of them still follow a frame-centric paradigm and apply similar representations to retained content regardless of its importance. This makes it difficult to preserve both high-fidelity visual evidence and broad temporal coverage. To address this issue, we propose Q-Fold, a tr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2606.12125","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2026-06-10T14:19:15Z","cross_cats_sorted":[],"title_canon_sha256":"3161cd6bccfc3ce144150f48dc72e6de8e2a94ce8aba8da2b381fcd2866b4409","abstract_canon_sha256":"7178dfb4800f1884b1755d986a35a312c50771bbd967573931f5187f87298fff"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-11T01:10:49.703265Z","signature_b64":"at/ls2mWGJCTuG8V/QFN1BwLAzI20y1D1CHJeQ2SGmqqfRuRwzPmpnvXHjmAETlhBKWpsKqAlCXbRzzo9guvCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e009972f09c05a3b16c4bce2a9ee070ecf4466f043e7d4b4e0a64f6b95d0255f","last_reissued_at":"2026-06-11T01:10:49.702567Z","signature_status":"signed_v1","first_computed_at":"2026-06-11T01:10:49.702567Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Q-Fold: Query-Aware Focus-Context Spatio-Temporal Folding for Long Video Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Biao Tang, Chenqiang Gao, Jingyi Yuan, Shuxiang Gou, Xu Chen, Yuhan Zhang","submitted_at":"2026-06-10T14:19:15Z","abstract_excerpt":"Long-video understanding remains challenging for multimodal large language models, because temporally extended videos often contain thousands of frames and are therefore expensive to process exhaustively. Existing methods usually construct compact visual inputs from long videos under a limited visual budget. However, most of them still follow a frame-centric paradigm and apply similar representations to retained content regardless of its importance. This makes it difficult to preserve both high-fidelity visual evidence and broad temporal coverage. To address this issue, we propose Q-Fold, a tr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2606.12125","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2606.12125/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2606.12125","created_at":"2026-06-11T01:10:49.702677+00:00"},{"alias_kind":"arxiv_version","alias_value":"2606.12125v1","created_at":"2026-06-11T01:10:49.702677+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2606.12125","created_at":"2026-06-11T01:10:49.702677+00:00"},{"alias_kind":"pith_short_12","alias_value":"4AEZOLYJYBND","created_at":"2026-06-11T01:10:49.702677+00:00"},{"alias_kind":"pith_short_16","alias_value":"4AEZOLYJYBNDWFWE","created_at":"2026-06-11T01:10:49.702677+00:00"},{"alias_kind":"pith_short_8","alias_value":"4AEZOLYJ","created_at":"2026-06-11T01:10:49.702677+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4AEZOLYJYBNDWFWEXTRKT3QHB3","json":"https://pith.science/pith/4AEZOLYJYBNDWFWEXTRKT3QHB3.json","graph_json":"https://pith.science/api/pith-number/4AEZOLYJYBNDWFWEXTRKT3QHB3/graph.json","events_json":"https://pith.science/api/pith-number/4AEZOLYJYBNDWFWEXTRKT3QHB3/events.json","paper":"https://pith.science/paper/4AEZOLYJ"},"agent_actions":{"view_html":"https://pith.science/pith/4AEZOLYJYBNDWFWEXTRKT3QHB3","download_json":"https://pith.science/pith/4AEZOLYJYBNDWFWEXTRKT3QHB3.json","view_paper":"https://pith.science/paper/4AEZOLYJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2606.12125&json=true","fetch_graph":"https://pith.science/api/pith-number/4AEZOLYJYBNDWFWEXTRKT3QHB3/graph.json","fetch_events":"https://pith.science/api/pith-number/4AEZOLYJYBNDWFWEXTRKT3QHB3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4AEZOLYJYBNDWFWEXTRKT3QHB3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4AEZOLYJYBNDWFWEXTRKT3QHB3/action/storage_attestation","attest_author":"https://pith.science/pith/4AEZOLYJYBNDWFWEXTRKT3QHB3/action/author_attestation","sign_citation":"https://pith.science/pith/4AEZOLYJYBNDWFWEXTRKT3QHB3/action/citation_signature","submit_replication":"https://pith.science/pith/4AEZOLYJYBNDWFWEXTRKT3QHB3/action/replication_record"}},"created_at":"2026-06-11T01:10:49.702677+00:00","updated_at":"2026-06-11T01:10:49.702677+00:00"}