{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:RLQYMBRASVHBF4UE45JXLOLGN2","short_pith_number":"pith:RLQYMBRA","schema_version":"1.0","canonical_sha256":"8ae1860620954e12f284e75375b9666e9914ae2ae3a2213ccc79a1f654584d25","source":{"kind":"arxiv","id":"2602.15257","version":3},"attestation_state":"computed","paper":{"title":"How to Train Your Long-Context Visual Document Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Austin Veselka","submitted_at":"2026-02-16T23:26:51Z","abstract_excerpt":"We present the first comprehensive, large-scale study of training long-context vision language models up to 344K context, targeting long-document visual question answering with measured transfer to long-context text. While several such strong are open-weight, namely Qwen3 VL and GLM 4.5/6V, their training recipes and data pipelines are not reproducible. We systematically study continued pretraining, supervised finetuning, and preference optimization for 24B and 32B parameter models, backed by extensive LC evaluations and ablations to bridge this gap, and achieve state-of-the-art performance on"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2602.15257","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2026-02-16T23:26:51Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"fdc11a4d8416b4d056e296952519b4313cc63f172262736df3affcbc260a2dbc","abstract_canon_sha256":"60706263ddb6c0808a285e0510970d1f650bb4c96f5f9a64a6661155255b8df6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-30T02:18:06.856484Z","signature_b64":"cPrW9rBHEikmUByN/yShHaLPiT5hyYyxc1dOQ8wtVZKbUrEiwD5NHWa2DwenphnZIJP3Fq+dVpRELYHTXEsmCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8ae1860620954e12f284e75375b9666e9914ae2ae3a2213ccc79a1f654584d25","last_reissued_at":"2026-06-30T02:18:06.855670Z","signature_status":"signed_v1","first_computed_at":"2026-06-30T02:18:06.855670Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How to Train Your Long-Context Visual Document Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Austin Veselka","submitted_at":"2026-02-16T23:26:51Z","abstract_excerpt":"We present the first comprehensive, large-scale study of training long-context vision language models up to 344K context, targeting long-document visual question answering with measured transfer to long-context text. While several such strong are open-weight, namely Qwen3 VL and GLM 4.5/6V, their training recipes and data pipelines are not reproducible. We systematically study continued pretraining, supervised finetuning, and preference optimization for 24B and 32B parameter models, backed by extensive LC evaluations and ablations to bridge this gap, and achieve state-of-the-art performance on"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2602.15257","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2602.15257/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2602.15257","created_at":"2026-06-30T02:18:06.855770+00:00"},{"alias_kind":"arxiv_version","alias_value":"2602.15257v3","created_at":"2026-06-30T02:18:06.855770+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2602.15257","created_at":"2026-06-30T02:18:06.855770+00:00"},{"alias_kind":"pith_short_12","alias_value":"RLQYMBRASVHB","created_at":"2026-06-30T02:18:06.855770+00:00"},{"alias_kind":"pith_short_16","alias_value":"RLQYMBRASVHBF4UE","created_at":"2026-06-30T02:18:06.855770+00:00"},{"alias_kind":"pith_short_8","alias_value":"RLQYMBRA","created_at":"2026-06-30T02:18:06.855770+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2605.13831","citing_title":"Training Long-Context Vision-Language Models Effectively with Generalization Beyond 128K Context","ref_index":38,"is_internal_anchor":true},{"citing_arxiv_id":"2604.02371","citing_title":"Internalized Reasoning for Long-Context Visual Document Understanding","ref_index":45,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RLQYMBRASVHBF4UE45JXLOLGN2","json":"https://pith.science/pith/RLQYMBRASVHBF4UE45JXLOLGN2.json","graph_json":"https://pith.science/api/pith-number/RLQYMBRASVHBF4UE45JXLOLGN2/graph.json","events_json":"https://pith.science/api/pith-number/RLQYMBRASVHBF4UE45JXLOLGN2/events.json","paper":"https://pith.science/paper/RLQYMBRA"},"agent_actions":{"view_html":"https://pith.science/pith/RLQYMBRASVHBF4UE45JXLOLGN2","download_json":"https://pith.science/pith/RLQYMBRASVHBF4UE45JXLOLGN2.json","view_paper":"https://pith.science/paper/RLQYMBRA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2602.15257&json=true","fetch_graph":"https://pith.science/api/pith-number/RLQYMBRASVHBF4UE45JXLOLGN2/graph.json","fetch_events":"https://pith.science/api/pith-number/RLQYMBRASVHBF4UE45JXLOLGN2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RLQYMBRASVHBF4UE45JXLOLGN2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RLQYMBRASVHBF4UE45JXLOLGN2/action/storage_attestation","attest_author":"https://pith.science/pith/RLQYMBRASVHBF4UE45JXLOLGN2/action/author_attestation","sign_citation":"https://pith.science/pith/RLQYMBRASVHBF4UE45JXLOLGN2/action/citation_signature","submit_replication":"https://pith.science/pith/RLQYMBRASVHBF4UE45JXLOLGN2/action/replication_record"}},"created_at":"2026-06-30T02:18:06.855770+00:00","updated_at":"2026-06-30T02:18:06.855770+00:00"}