{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:YVTCF3NOJ3VFRZZI4IFKB6ZXE3","short_pith_number":"pith:YVTCF3NO","schema_version":"1.0","canonical_sha256":"c56622edae4eea58e728e20aa0fb3726d64ea2350b8a64e8eb6914ae5b070e5f","source":{"kind":"arxiv","id":"2503.18908","version":1},"attestation_state":"computed","paper":{"title":"FFN Fusion: Rethinking Sequential Computation in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Akhiad Bercovich, Amnon Geifman, Ehud Karpas, Elad Segal, Ido Galil, Ido Shahaf, Itamar Schen, Itay Levy, Izhak Golan, Mohammad Dabbah, Najeeb Nabwani, Omri Puny, Oren Tropp, Ran El-Yaniv, Ran Zilberstein, Tomer Ronen, Yonatan Geifman, Zach Moshe","submitted_at":"2025-03-24T17:20:35Z","abstract_excerpt":"We introduce FFN Fusion, an architectural optimization technique that reduces sequential computation in large language models by identifying and exploiting natural opportunities for parallelization. Our key insight is that sequences of Feed-Forward Network (FFN) layers, particularly those remaining after the removal of specific attention layers, can often be parallelized with minimal accuracy impact. We develop a principled methodology for identifying and fusing such sequences, transforming them into parallel operations that significantly reduce inference latency while preserving model behavio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.18908","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-03-24T17:20:35Z","cross_cats_sorted":[],"title_canon_sha256":"651cd293f29d243f85d9cbb85078c1969b020537fd41c07bbda808cde5a1ee0d","abstract_canon_sha256":"86bdfd85ac46cd718722900777bff3a59deee81505ab2380ed5bfe7ba29b09ae"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:38:28.068582Z","signature_b64":"EEanhLhI+onK5KD/LX0D7n/nXQbIAZj3cK4qqv8fS2fM6T4W0XElB964EDIcprRP/4oJ+m5mZ2quQJP3d7TNBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c56622edae4eea58e728e20aa0fb3726d64ea2350b8a64e8eb6914ae5b070e5f","last_reissued_at":"2026-07-05T10:38:28.067996Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:38:28.067996Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FFN Fusion: Rethinking Sequential Computation in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Akhiad Bercovich, Amnon Geifman, Ehud Karpas, Elad Segal, Ido Galil, Ido Shahaf, Itamar Schen, Itay Levy, Izhak Golan, Mohammad Dabbah, Najeeb Nabwani, Omri Puny, Oren Tropp, Ran El-Yaniv, Ran Zilberstein, Tomer Ronen, Yonatan Geifman, Zach Moshe","submitted_at":"2025-03-24T17:20:35Z","abstract_excerpt":"We introduce FFN Fusion, an architectural optimization technique that reduces sequential computation in large language models by identifying and exploiting natural opportunities for parallelization. Our key insight is that sequences of Feed-Forward Network (FFN) layers, particularly those remaining after the removal of specific attention layers, can often be parallelized with minimal accuracy impact. We develop a principled methodology for identifying and fusing such sequences, transforming them into parallel operations that significantly reduce inference latency while preserving model behavio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.18908","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.18908/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.18908","created_at":"2026-07-05T10:38:28.068066+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.18908v1","created_at":"2026-07-05T10:38:28.068066+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.18908","created_at":"2026-07-05T10:38:28.068066+00:00"},{"alias_kind":"pith_short_12","alias_value":"YVTCF3NOJ3VF","created_at":"2026-07-05T10:38:28.068066+00:00"},{"alias_kind":"pith_short_16","alias_value":"YVTCF3NOJ3VFRZZI","created_at":"2026-07-05T10:38:28.068066+00:00"},{"alias_kind":"pith_short_8","alias_value":"YVTCF3NO","created_at":"2026-07-05T10:38:28.068066+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.05340","citing_title":"Exploring Diffusion Transformer Designs via Grafting","ref_index":49,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YVTCF3NOJ3VFRZZI4IFKB6ZXE3","json":"https://pith.science/pith/YVTCF3NOJ3VFRZZI4IFKB6ZXE3.json","graph_json":"https://pith.science/api/pith-number/YVTCF3NOJ3VFRZZI4IFKB6ZXE3/graph.json","events_json":"https://pith.science/api/pith-number/YVTCF3NOJ3VFRZZI4IFKB6ZXE3/events.json","paper":"https://pith.science/paper/YVTCF3NO"},"agent_actions":{"view_html":"https://pith.science/pith/YVTCF3NOJ3VFRZZI4IFKB6ZXE3","download_json":"https://pith.science/pith/YVTCF3NOJ3VFRZZI4IFKB6ZXE3.json","view_paper":"https://pith.science/paper/YVTCF3NO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.18908&json=true","fetch_graph":"https://pith.science/api/pith-number/YVTCF3NOJ3VFRZZI4IFKB6ZXE3/graph.json","fetch_events":"https://pith.science/api/pith-number/YVTCF3NOJ3VFRZZI4IFKB6ZXE3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YVTCF3NOJ3VFRZZI4IFKB6ZXE3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YVTCF3NOJ3VFRZZI4IFKB6ZXE3/action/storage_attestation","attest_author":"https://pith.science/pith/YVTCF3NOJ3VFRZZI4IFKB6ZXE3/action/author_attestation","sign_citation":"https://pith.science/pith/YVTCF3NOJ3VFRZZI4IFKB6ZXE3/action/citation_signature","submit_replication":"https://pith.science/pith/YVTCF3NOJ3VFRZZI4IFKB6ZXE3/action/replication_record"}},"created_at":"2026-07-05T10:38:28.068066+00:00","updated_at":"2026-07-05T10:38:28.068066+00:00"}