{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:YGOEO3DG74YUIAF64WIPOORZ3K","short_pith_number":"pith:YGOEO3DG","schema_version":"1.0","canonical_sha256":"c19c476c66ff314400bee590f73a39da902d5e9b2a7fdc726c432d5fb7fa77ff","source":{"kind":"arxiv","id":"2502.18845","version":2},"attestation_state":"computed","paper":{"title":"Sliding Window Attention Training for Efficient Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Derong Xu, Tong Xu, Wentao Song, Xiangyu Zhao, Xian Wu, Xuetao Wei, Yefeng Zheng, Yejing Wang, Yingying Zhang, Zichuan Fu","submitted_at":"2025-02-26T05:31:44Z","abstract_excerpt":"Recent advances in transformer-based Large Language Models (LLMs) have demonstrated remarkable capabilities across various tasks. However, their quadratic computational complexity concerning sequence length remains a significant bottleneck for processing long documents. As a result, many efforts like sparse attention and state space models have been proposed to improve the efficiency of LLMs over long sequences. Though effective, these approaches compromise the performance or introduce structural complexity. This calls for a simple yet efficient model that preserves the fundamental Transformer"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.18845","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-26T05:31:44Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"78bbc3704c590d559fbc3ae58369b00628c30ab86af36e57ae4a75b091109504","abstract_canon_sha256":"6315d7da929349cbaf37c1fad1aea396e2780697f10886061ea4bde795533b24"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:15:43.491211Z","signature_b64":"gl1+824W1CnZxuH0fHNEjGfkH1gKLfsid/H2EXiy4ujsdmG0JZCqqMrh+W2u0xQEe9GbTDzJJHJhyjwmi5kcDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c19c476c66ff314400bee590f73a39da902d5e9b2a7fdc726c432d5fb7fa77ff","last_reissued_at":"2026-07-05T11:15:43.490723Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:15:43.490723Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Sliding Window Attention Training for Efficient Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Derong Xu, Tong Xu, Wentao Song, Xiangyu Zhao, Xian Wu, Xuetao Wei, Yefeng Zheng, Yejing Wang, Yingying Zhang, Zichuan Fu","submitted_at":"2025-02-26T05:31:44Z","abstract_excerpt":"Recent advances in transformer-based Large Language Models (LLMs) have demonstrated remarkable capabilities across various tasks. However, their quadratic computational complexity concerning sequence length remains a significant bottleneck for processing long documents. As a result, many efforts like sparse attention and state space models have been proposed to improve the efficiency of LLMs over long sequences. Though effective, these approaches compromise the performance or introduce structural complexity. This calls for a simple yet efficient model that preserves the fundamental Transformer"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.18845","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.18845/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.18845","created_at":"2026-07-05T11:15:43.490784+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.18845v2","created_at":"2026-07-05T11:15:43.490784+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.18845","created_at":"2026-07-05T11:15:43.490784+00:00"},{"alias_kind":"pith_short_12","alias_value":"YGOEO3DG74YU","created_at":"2026-07-05T11:15:43.490784+00:00"},{"alias_kind":"pith_short_16","alias_value":"YGOEO3DG74YUIAF6","created_at":"2026-07-05T11:15:43.490784+00:00"},{"alias_kind":"pith_short_8","alias_value":"YGOEO3DG","created_at":"2026-07-05T11:15:43.490784+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.04569","citing_title":"LIVEditor-14B: Lightning Unified Video Editing via In-Context Sparse Attention","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28831","citing_title":"HARD-KV: Head-Adaptive Regularization for Decoding-time KV Compression","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2504.02181","citing_title":"A Survey of Scaling in Large Language Model Reasoning","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2505.03258","citing_title":"IAFormer: Interaction-Aware Transformer network for collider data analysis","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07731","citing_title":"Benchmarking EngGPT2-16B-A3B against Comparable Italian and International Open-source LLMs","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2510.26083","citing_title":"Nirvana: A Specialized Generalist Model With Task-Aware Memory Mechanism","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11744","citing_title":"Training-Inference Consistent Segmented Execution for Long-Context LLMs","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10098","citing_title":"Attention Sink in Transformers: A Survey on Utilization, Interpretation, and Mitigation","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07731","citing_title":"Benchmarking EngGPT2-16B-A3B against Comparable Italian and International Open-source LLMs","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04451","citing_title":"Beyond Few-Step Inference: Accelerating Video Diffusion Transformer Model Serving with Inter-Request Caching Reuse","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YGOEO3DG74YUIAF64WIPOORZ3K","json":"https://pith.science/pith/YGOEO3DG74YUIAF64WIPOORZ3K.json","graph_json":"https://pith.science/api/pith-number/YGOEO3DG74YUIAF64WIPOORZ3K/graph.json","events_json":"https://pith.science/api/pith-number/YGOEO3DG74YUIAF64WIPOORZ3K/events.json","paper":"https://pith.science/paper/YGOEO3DG"},"agent_actions":{"view_html":"https://pith.science/pith/YGOEO3DG74YUIAF64WIPOORZ3K","download_json":"https://pith.science/pith/YGOEO3DG74YUIAF64WIPOORZ3K.json","view_paper":"https://pith.science/paper/YGOEO3DG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.18845&json=true","fetch_graph":"https://pith.science/api/pith-number/YGOEO3DG74YUIAF64WIPOORZ3K/graph.json","fetch_events":"https://pith.science/api/pith-number/YGOEO3DG74YUIAF64WIPOORZ3K/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YGOEO3DG74YUIAF64WIPOORZ3K/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YGOEO3DG74YUIAF64WIPOORZ3K/action/storage_attestation","attest_author":"https://pith.science/pith/YGOEO3DG74YUIAF64WIPOORZ3K/action/author_attestation","sign_citation":"https://pith.science/pith/YGOEO3DG74YUIAF64WIPOORZ3K/action/citation_signature","submit_replication":"https://pith.science/pith/YGOEO3DG74YUIAF64WIPOORZ3K/action/replication_record"}},"created_at":"2026-07-05T11:15:43.490784+00:00","updated_at":"2026-07-05T11:15:43.490784+00:00"}