{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:Z3D2D42UFXOFBBTPKOLZPWEQ3W","short_pith_number":"pith:Z3D2D42U","schema_version":"1.0","canonical_sha256":"cec7a1f3542ddc50866f539797d890ddbbba9633b51698e90c5fe7dcdfc55941","source":{"kind":"arxiv","id":"2501.16182","version":1},"attestation_state":"computed","paper":{"title":"The Linear Attention Resurrection in Vision Transformer","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Chuanyang Zheng","submitted_at":"2025-01-27T16:29:17Z","abstract_excerpt":"Vision Transformers (ViTs) have recently taken computer vision by storm. However, the softmax attention underlying ViTs comes with a quadratic complexity in time and memory, hindering the application of ViTs to high-resolution images. We revisit the attention design and propose a linear attention method to address the limitation, which doesn't sacrifice ViT's core advantage of capturing global representation like existing methods (e.g. local window attention of Swin). We further investigate the key difference between linear attention and softmax attention. Our empirical results suggest that li"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.16182","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-01-27T16:29:17Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"56e4d84d4e6bf517627f5568d0efc11c89d0b99c9b30420c3faad403b0419a1f","abstract_canon_sha256":"c1fa7cc3ac20430fc746b1c8d3cbb5322a172e4d8146de87e5047af50eec4660"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:14:25.864996Z","signature_b64":"ErsXxfPz50Qi91n3ndPCZZnjUlEzcRmkGvp7ZQvuyJqUycP/A8W4ht59/opVCU8tCCK580uUyJ5li4UBR+4SBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cec7a1f3542ddc50866f539797d890ddbbba9633b51698e90c5fe7dcdfc55941","last_reissued_at":"2026-07-05T10:14:25.864504Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:14:25.864504Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Linear Attention Resurrection in Vision Transformer","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Chuanyang Zheng","submitted_at":"2025-01-27T16:29:17Z","abstract_excerpt":"Vision Transformers (ViTs) have recently taken computer vision by storm. However, the softmax attention underlying ViTs comes with a quadratic complexity in time and memory, hindering the application of ViTs to high-resolution images. We revisit the attention design and propose a linear attention method to address the limitation, which doesn't sacrifice ViT's core advantage of capturing global representation like existing methods (e.g. local window attention of Swin). We further investigate the key difference between linear attention and softmax attention. Our empirical results suggest that li"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.16182","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.16182/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.16182","created_at":"2026-07-05T10:14:25.864564+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.16182v1","created_at":"2026-07-05T10:14:25.864564+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.16182","created_at":"2026-07-05T10:14:25.864564+00:00"},{"alias_kind":"pith_short_12","alias_value":"Z3D2D42UFXOF","created_at":"2026-07-05T10:14:25.864564+00:00"},{"alias_kind":"pith_short_16","alias_value":"Z3D2D42UFXOFBBTP","created_at":"2026-07-05T10:14:25.864564+00:00"},{"alias_kind":"pith_short_8","alias_value":"Z3D2D42U","created_at":"2026-07-05T10:14:25.864564+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.18848","citing_title":"Exact Linear Attention","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25365","citing_title":"Quantum Parameterized Self-Attention Network for Image Classification","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18848","citing_title":"Exact Linear Attention","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23798","citing_title":"ELSA: Exact Linear-Scan Attention for Fast and Memory-Light Vision Transformers","ref_index":49,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Z3D2D42UFXOFBBTPKOLZPWEQ3W","json":"https://pith.science/pith/Z3D2D42UFXOFBBTPKOLZPWEQ3W.json","graph_json":"https://pith.science/api/pith-number/Z3D2D42UFXOFBBTPKOLZPWEQ3W/graph.json","events_json":"https://pith.science/api/pith-number/Z3D2D42UFXOFBBTPKOLZPWEQ3W/events.json","paper":"https://pith.science/paper/Z3D2D42U"},"agent_actions":{"view_html":"https://pith.science/pith/Z3D2D42UFXOFBBTPKOLZPWEQ3W","download_json":"https://pith.science/pith/Z3D2D42UFXOFBBTPKOLZPWEQ3W.json","view_paper":"https://pith.science/paper/Z3D2D42U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.16182&json=true","fetch_graph":"https://pith.science/api/pith-number/Z3D2D42UFXOFBBTPKOLZPWEQ3W/graph.json","fetch_events":"https://pith.science/api/pith-number/Z3D2D42UFXOFBBTPKOLZPWEQ3W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Z3D2D42UFXOFBBTPKOLZPWEQ3W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Z3D2D42UFXOFBBTPKOLZPWEQ3W/action/storage_attestation","attest_author":"https://pith.science/pith/Z3D2D42UFXOFBBTPKOLZPWEQ3W/action/author_attestation","sign_citation":"https://pith.science/pith/Z3D2D42UFXOFBBTPKOLZPWEQ3W/action/citation_signature","submit_replication":"https://pith.science/pith/Z3D2D42UFXOFBBTPKOLZPWEQ3W/action/replication_record"}},"created_at":"2026-07-05T10:14:25.864564+00:00","updated_at":"2026-07-05T10:14:25.864564+00:00"}