{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BYWUS5LKKQXKY6BV67YS2TPZPD","short_pith_number":"pith:BYWUS5LK","schema_version":"1.0","canonical_sha256":"0e2d49756a542eac7835f7f12d4df978e0574ed227bf4b7fa05defaceecd6a57","source":{"kind":"arxiv","id":"2402.04347","version":1},"attestation_state":"computed","paper":{"title":"The Hedgehog & the Porcupine: Expressive Linear Attentions with Softmax Mimicry","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Christopher R\\'e, Hermann Kumbong, Kush Bhatia, Michael Zhang","submitted_at":"2024-02-06T19:31:26Z","abstract_excerpt":"Linear attentions have shown potential for improving Transformer efficiency, reducing attention's quadratic complexity to linear in sequence length. This holds exciting promise for (1) training linear Transformers from scratch, (2) \"finetuned-conversion\" of task-specific Transformers into linear versions that recover task performance, and (3) \"pretrained-conversion\" of Transformers such as large language models into linear versions finetunable on downstream tasks. However, linear attentions often underperform standard softmax attention in quality. To close this performance gap, we find prior l"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.04347","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-02-06T19:31:26Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"1e114d3b3de7395670fa7b2c6272d3cb9e61d27aeb357871b64cec0fbad6b227","abstract_canon_sha256":"a1702a2d3924ec9c8333c8e729829c6fc886a340fb095e62c9c492820ff5ace0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:42:07.336270Z","signature_b64":"2gRlcf2e+4rPtd7cRb8rznLX9d3eId6utPjOu1yAUOg8x1Pnc5A01NUm+tCFsgQskx7C3fp7FjZE/ZcLV7OeCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0e2d49756a542eac7835f7f12d4df978e0574ed227bf4b7fa05defaceecd6a57","last_reissued_at":"2026-07-05T07:42:07.335821Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:42:07.335821Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Hedgehog & the Porcupine: Expressive Linear Attentions with Softmax Mimicry","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Christopher R\\'e, Hermann Kumbong, Kush Bhatia, Michael Zhang","submitted_at":"2024-02-06T19:31:26Z","abstract_excerpt":"Linear attentions have shown potential for improving Transformer efficiency, reducing attention's quadratic complexity to linear in sequence length. This holds exciting promise for (1) training linear Transformers from scratch, (2) \"finetuned-conversion\" of task-specific Transformers into linear versions that recover task performance, and (3) \"pretrained-conversion\" of Transformers such as large language models into linear versions finetunable on downstream tasks. However, linear attentions often underperform standard softmax attention in quality. To close this performance gap, we find prior l"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.04347","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.04347/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.04347","created_at":"2026-07-05T07:42:07.335886+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.04347v1","created_at":"2026-07-05T07:42:07.335886+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.04347","created_at":"2026-07-05T07:42:07.335886+00:00"},{"alias_kind":"pith_short_12","alias_value":"BYWUS5LKKQXK","created_at":"2026-07-05T07:42:07.335886+00:00"},{"alias_kind":"pith_short_16","alias_value":"BYWUS5LKKQXKY6BV","created_at":"2026-07-05T07:42:07.335886+00:00"},{"alias_kind":"pith_short_8","alias_value":"BYWUS5LK","created_at":"2026-07-05T07:42:07.335886+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07706","citing_title":"The Key to Going Linear: Analysis-Driven Transformer Linearization","ref_index":27,"is_internal_anchor":true},{"citing_arxiv_id":"2606.09862","citing_title":"Blurry Window Attention","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02772","citing_title":"Linearizing Vision Transformer with Test-Time Training","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30562","citing_title":"Morphing into Hybrid Attention Models","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31559","citing_title":"Functional Attention: From Pairwise Affinities to Functional Correspondences","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2507.09025","citing_title":"Lizard: An Efficient Linearization Framework for Large Language Models","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2603.11680","citing_title":"UCAN: Unified Convolutional Attention Network for Expansive Receptive Fields in Lightweight Super-Resolution","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14191","citing_title":"Attention to Mamba: A Recipe for Cross-Architecture Distillation","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12491","citing_title":"Elastic Attention Cores for Scalable Vision Transformers","ref_index":55,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BYWUS5LKKQXKY6BV67YS2TPZPD","json":"https://pith.science/pith/BYWUS5LKKQXKY6BV67YS2TPZPD.json","graph_json":"https://pith.science/api/pith-number/BYWUS5LKKQXKY6BV67YS2TPZPD/graph.json","events_json":"https://pith.science/api/pith-number/BYWUS5LKKQXKY6BV67YS2TPZPD/events.json","paper":"https://pith.science/paper/BYWUS5LK"},"agent_actions":{"view_html":"https://pith.science/pith/BYWUS5LKKQXKY6BV67YS2TPZPD","download_json":"https://pith.science/pith/BYWUS5LKKQXKY6BV67YS2TPZPD.json","view_paper":"https://pith.science/paper/BYWUS5LK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.04347&json=true","fetch_graph":"https://pith.science/api/pith-number/BYWUS5LKKQXKY6BV67YS2TPZPD/graph.json","fetch_events":"https://pith.science/api/pith-number/BYWUS5LKKQXKY6BV67YS2TPZPD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BYWUS5LKKQXKY6BV67YS2TPZPD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BYWUS5LKKQXKY6BV67YS2TPZPD/action/storage_attestation","attest_author":"https://pith.science/pith/BYWUS5LKKQXKY6BV67YS2TPZPD/action/author_attestation","sign_citation":"https://pith.science/pith/BYWUS5LKKQXKY6BV67YS2TPZPD/action/citation_signature","submit_replication":"https://pith.science/pith/BYWUS5LKKQXKY6BV67YS2TPZPD/action/replication_record"}},"created_at":"2026-07-05T07:42:07.335886+00:00","updated_at":"2026-07-05T07:42:07.335886+00:00"}