{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YHDYCGGH5PTP5DS3CGDBJJK6YS","short_pith_number":"pith:YHDYCGGH","schema_version":"1.0","canonical_sha256":"c1c78118c7ebe6fe8e5b118614a55ec49d84e991c0c94c61ef7106dabb2fa6cf","source":{"kind":"arxiv","id":"2403.08845","version":2},"attestation_state":"computed","paper":{"title":"Bifurcated Attention: Accelerating Massively Parallel Decoding with Shared Prefixes in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Ben Athiwaratkun, Bing Xiang, Haifeng Qian, Hantian Ding, Jiacheng Guo, Jun Wang, Liangfu Chen, Parminder Bhatia, Qing Sun, Ramesh Nallapati, Sanjay Krishna Gouda, Sudipta Sengupta, Sujan Kumar Gonugondla","submitted_at":"2024-03-13T16:30:57Z","abstract_excerpt":"This study introduces bifurcated attention, a method designed to enhance language model inference in shared-context batch decoding scenarios. Our approach addresses the challenge of redundant memory IO costs, a critical factor contributing to latency in high batch sizes and extended context lengths. Bifurcated attention achieves this by strategically dividing the attention mechanism during incremental decoding into two separate GEMM operations: one focusing on the KV cache from prefill, and another on the decoding process itself. While maintaining the computational load (FLOPs) of standard att"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.08845","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-03-13T16:30:57Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"5000ff13e11f8fb1b94b277372f4183c8a5471bb5d4606933ba44f4b07239c91","abstract_canon_sha256":"df328929350ca7aa32a7100894a8dce211fc2d60d9d3cfdcb16810a8083c816e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:42:53.866540Z","signature_b64":"U5hKYRi5W1TVbXBLkPPb9Oc7U3ErKK88BBjlF9kx8qmt8HAZtSdrz4WW2kyaSVrrvF4D3YsXi20r2UIh6WeRDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c1c78118c7ebe6fe8e5b118614a55ec49d84e991c0c94c61ef7106dabb2fa6cf","last_reissued_at":"2026-07-05T08:42:53.865993Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:42:53.865993Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Bifurcated Attention: Accelerating Massively Parallel Decoding with Shared Prefixes in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Ben Athiwaratkun, Bing Xiang, Haifeng Qian, Hantian Ding, Jiacheng Guo, Jun Wang, Liangfu Chen, Parminder Bhatia, Qing Sun, Ramesh Nallapati, Sanjay Krishna Gouda, Sudipta Sengupta, Sujan Kumar Gonugondla","submitted_at":"2024-03-13T16:30:57Z","abstract_excerpt":"This study introduces bifurcated attention, a method designed to enhance language model inference in shared-context batch decoding scenarios. Our approach addresses the challenge of redundant memory IO costs, a critical factor contributing to latency in high batch sizes and extended context lengths. Bifurcated attention achieves this by strategically dividing the attention mechanism during incremental decoding into two separate GEMM operations: one focusing on the KV cache from prefill, and another on the decoding process itself. While maintaining the computational load (FLOPs) of standard att"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.08845","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.08845/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.08845","created_at":"2026-07-05T08:42:53.866055+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.08845v2","created_at":"2026-07-05T08:42:53.866055+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.08845","created_at":"2026-07-05T08:42:53.866055+00:00"},{"alias_kind":"pith_short_12","alias_value":"YHDYCGGH5PTP","created_at":"2026-07-05T08:42:53.866055+00:00"},{"alias_kind":"pith_short_16","alias_value":"YHDYCGGH5PTP5DS3","created_at":"2026-07-05T08:42:53.866055+00:00"},{"alias_kind":"pith_short_8","alias_value":"YHDYCGGH","created_at":"2026-07-05T08:42:53.866055+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.15422","citing_title":"DualKV: Shared-Prompt Flash Attention for Efficient RL Training with Large Rollouts and Long Contexts","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15422","citing_title":"DualKV: Shared-Prompt Flash Attention for Efficient RL Training with Large Rollouts and Long Contexts","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2507.13334","citing_title":"A Survey of Context Engineering for Large Language Models","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2407.21787","citing_title":"Large Language Monkeys: Scaling Inference Compute with Repeated Sampling","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YHDYCGGH5PTP5DS3CGDBJJK6YS","json":"https://pith.science/pith/YHDYCGGH5PTP5DS3CGDBJJK6YS.json","graph_json":"https://pith.science/api/pith-number/YHDYCGGH5PTP5DS3CGDBJJK6YS/graph.json","events_json":"https://pith.science/api/pith-number/YHDYCGGH5PTP5DS3CGDBJJK6YS/events.json","paper":"https://pith.science/paper/YHDYCGGH"},"agent_actions":{"view_html":"https://pith.science/pith/YHDYCGGH5PTP5DS3CGDBJJK6YS","download_json":"https://pith.science/pith/YHDYCGGH5PTP5DS3CGDBJJK6YS.json","view_paper":"https://pith.science/paper/YHDYCGGH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.08845&json=true","fetch_graph":"https://pith.science/api/pith-number/YHDYCGGH5PTP5DS3CGDBJJK6YS/graph.json","fetch_events":"https://pith.science/api/pith-number/YHDYCGGH5PTP5DS3CGDBJJK6YS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YHDYCGGH5PTP5DS3CGDBJJK6YS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YHDYCGGH5PTP5DS3CGDBJJK6YS/action/storage_attestation","attest_author":"https://pith.science/pith/YHDYCGGH5PTP5DS3CGDBJJK6YS/action/author_attestation","sign_citation":"https://pith.science/pith/YHDYCGGH5PTP5DS3CGDBJJK6YS/action/citation_signature","submit_replication":"https://pith.science/pith/YHDYCGGH5PTP5DS3CGDBJJK6YS/action/replication_record"}},"created_at":"2026-07-05T08:42:53.866055+00:00","updated_at":"2026-07-05T08:42:53.866055+00:00"}