{"as_of":"2026-08-16T06:53:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a8d2bc433d34277188c869ca46c633b4e1f0bcc3743ad0c69363e1abdff35003","coverage":[{"denominator":89,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":89,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T13:46:01.321581Z","state":"measured"},{"denominator":89,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":89,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-16T06:30:59.297886+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2411.15982/citation-record","integrity":"/paper/2411.15982/integrity","json":"/paper/2411.15982/citation-record.json","paper":"/paper/2411.15982"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:00.916127Z","title":"Resq: Residual quantization for video perception,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.916127Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:0bd67e8ef098eb1e393cb258a104def6711bc5dc187a19c1cc5b49e5b1eb8f79","observation_id":"eb908304-59b4-4082-8b4e-8ca13af78c73","resolution":{"observed_at":"2026-08-12T13:46:00.916127Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:00.924633Z","title":"Bit-pragmatic deep neural network computing,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.924633Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:18c2fa9d657ab7e1b0e3542ec7b44a9fb9e763b439bc133c43fbb4e7ea6b77af","observation_id":"07b6e85e-10eb-4742-ad69-eb44c17f2a83","resolution":{"observed_at":"2026-08-12T13:46:00.924633Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:00.928808Z","title":"Explaining neural scaling laws,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.928808Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:43d96805e38cbc2ec20dffc061e95b293c577a915cadad80856371ebb9d1ba98","observation_id":"109156cc-ec88-4c3e-b740-788a6f30fc11","resolution":{"observed_at":"2026-08-12T13:46:00.928808Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:00.933063Z","title":"Longbench: A bilingual, multitask benchmark for long context understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.933063Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:623218d2755b80ca29bb661b7340adedab0494332d041276ed6ffadb9eab3889","observation_id":"f22e992c-6df8-4344-b35e-d24c47398f35","resolution":{"observed_at":"2026-08-12T13:46:00.933063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:00.937312Z","title":"Demystifying chatgpt: An in-depth survey of openai’s robust large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.937312Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:a16c5175750ecee07b0c9b27a07215c65defd88bff8156e08378d7c300d6c6f4","observation_id":"c2fbf310-6db7-48f4-b945-3546b592e458","resolution":{"observed_at":"2026-08-12T13:46:00.937312Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:00.941946Z","title":"Genus synthesis solution,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.941946Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:2d46f839660002ee5766208ec14c56f92c7687a24bd644f3b42773f1d055b555","observation_id":"512cd457-8d26-4023-b16d-7a04113f5a87","resolution":{"observed_at":"2026-08-12T13:46:00.941946Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:00.946229Z","title":"General purpose deep learning accelerator based on bit interleaving,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.946229Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:9960d0af2f74512b7963c6fe5a437c320b6de75b6cfe4c51d62400e2194fa0df","observation_id":"5c53dec2-f8da-4aba-a7c7-4e8f421615c9","resolution":{"observed_at":"2026-08-12T13:46:00.946229Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:00.950195Z","title":"Quip: 2-bit quanti- zation of large language models with guarantees,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.950195Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:8227b1fa5b1d852b83932e0a7d32f8cc200f3de4d0926b85cce1912c42116476","observation_id":"f864ed24-45c9-401f-ae54-5c8143377e09","resolution":{"observed_at":"2026-08-12T13:46:00.950195Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.11062","last_updated":"2025-05-19T06:20:01Z","snapshot_observed_at":"2026-08-13T19:22:45.524352Z","submitted_at":"2024-07-10T17:53:30Z","title":"EfficientQAT: Efficient Quantization-Aware Training for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.11062","snapshot_observed_at":"2026-08-12T13:46:00.954375Z","title":"Efficientqat: Efficient quantization-aware training for large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.954375Z"},"links":{"cited_paper":"/paper/2407.11062","citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:5872936f1e149aff067cde0dda2085cb985d1b9b5be4bbeaca3aff98f258b319","observation_id":"db8e462a-25cb-46fb-b2b5-1b5a23573c3e","resolution":{"observed_at":"2026-08-12T13:46:00.954375Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:00.959299Z","title":"Nacl: A general and effective kv cache eviction framework for llm at inference time,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.959299Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:9202a5e3ddc37753b9599855e25b51e3e6be3d8ff48bad46befaa595c00622a7","observation_id":"c08c524f-0c14-4d61-b5d6-f0409de0d6e1","resolution":{"observed_at":"2026-08-12T13:46:00.959299Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.683922Z","title":"Palm: Scaling language modeling with pathways,","venue":null,"work_id":"6989fed4-0dce-4c5d-bb3f-dcbca285a378","year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.963367Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:50a3f409d7545833efbac2a9b8f0e9555f9f2aa64c70c78253c97a15dfba5de1","observation_id":"3e140eb0-9c01-4fba-b3db-04b06150bd15","resolution":{"observed_at":"2026-08-12T13:46:02.688552Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.668947Z","title":"Vs-quant: Per-vector scaled quantization for accurate low-precision neural network inference,","venue":null,"work_id":"e6ec428c-42eb-49b0-96c3-2022726f2244","year":2021},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.967701Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:48670b9a96d18146cd37555832742363f7a77a6efa01874541e5c6e8478e683c","observation_id":"0e391695-d4ee-48c0-a897-a30f0f232543","resolution":{"observed_at":"2026-08-12T13:46:02.673582Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.655841Z","title":"Pushing the limits of narrow precision inferencing at cloud scale with microsoft floating point,","venue":null,"work_id":"d0ea9865-bddf-4562-9167-0e1caa06eadc","year":2020},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.971544Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:923b7086b144a0b74b8df1f9007983f6cf11c2e92154d7b39c3ea1f944cf6e58","observation_id":"cbfde174-f710-43b3-bc1e-c0588e604377","resolution":{"observed_at":"2026-08-12T13:46:02.660384Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.643539Z","title":"With shared microexponents, a little shifting goes a long way,","venue":null,"work_id":"a061b0c8-15bf-4e57-8ea9-fca663d4d90b","year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.976155Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:6b46470ccaa53057accf89e5e9fa130f2e016f4729dce96ae019864cb5123dc9","observation_id":"987c2874-d2d8-4184-a3d2-f795bbdf96e0","resolution":{"observed_at":"2026-08-12T13:46:02.647418Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.630764Z","title":"A timing-driven approach to synthesize fast barrel shifters,","venue":null,"work_id":"ef69b834-1e62-4204-a5da-ccda12395395","year":2008},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.980180Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:0c7eb0ee3eeb31001cc0e6bec08a18acdf55fe7fbce666a712b214279e2b5c8d","observation_id":"d0088e03-3677-4f8a-8050-e4d1ccb8efc8","resolution":{"observed_at":"2026-08-12T13:46:02.635399Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.617250Z","title":"Llm.int8(): 8- bit matrix multiplication for transformers at scale,","venue":null,"work_id":"cb4a5ad0-724f-4a4f-b6a3-1115c67304f7","year":2022},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.984391Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:8b2e017d33058f24e1dbb7a7acba0baa51013df99103e207bca82b99429e06a4","observation_id":"ec788826-bdd6-4922-b1b5-c725344a1534","resolution":{"observed_at":"2026-08-12T13:46:02.622334Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.600868Z","title":"The case for 4-bit precision: k- bit inference scaling laws,","venue":null,"work_id":"efd3de95-d187-4ca0-a777-561280b8b94a","year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.988162Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:613130fc81778a2a869fcf4f3ee7d62ae608590a6dd440c1da146778070b7d8f","observation_id":"8480b9a4-4201-4813-a0d8-68b85334e80b","resolution":{"observed_at":"2026-08-12T13:46:02.606453Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.582859Z","title":"Hawq: Hessian aware quantization of neural networks with mixed-precision,","venue":null,"work_id":"22b7c12a-b4fc-4b7a-9004-db0e82e44252","year":2019},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.992180Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:afaef8caf511ae34a091c947cf5429d571d763edfe2b8cef5e7d0704299c287b","observation_id":"00ecbf6f-8afa-4325-9fb0-89afa02558c9","resolution":{"observed_at":"2026-08-12T13:46:02.588365Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.566937Z","title":"Training dnns with hybrid block floating point,","venue":null,"work_id":"ce681c5a-2aeb-4e54-9149-18e733bb6c61","year":2018},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.996132Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:9105bbb8dce953ddd66016c528836b614b553757235844518cef82a9e8b57d81","observation_id":"e6510551-5932-40cc-b624-1173024cb859","resolution":{"observed_at":"2026-08-12T13:46:02.571178Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.553464Z","title":"Skvq: Sliding-window key and value cache quantization for large language models,","venue":null,"work_id":"dfeb37c8-4dea-43d7-ad8b-7bfca1ed6a54","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.000362Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:feafd48cae804d71f239a369bd8801dde9a37fced1b3198c77addd9d37b8db64","observation_id":"dce06533-ae2f-40b8-af4b-eab6c87e6b06","resolution":{"observed_at":"2026-08-12T13:46:02.557598Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.540840Z","title":"Extreme compression of large language models via additive quantization,","venue":null,"work_id":"90b2ed13-44fe-43ec-a28f-e7728661a0f0","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.004503Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:73f9af983b9d782fb8721f9090fb1c64dde493d7aeda1e992d79ced6834b20d2","observation_id":"178aaf15-301b-40f6-8b13-bc2fbe940d64","resolution":{"observed_at":"2026-08-12T13:46:02.545100Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.528329Z","title":"Reconfig- urable acceleration of 3d-cnns for human action recognition with block floating-point representation,","venue":null,"work_id":"7dfb0ec6-ff84-4830-af34-3cb65b129f72","year":2018},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.008439Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:686117a2118b068ddd958aceb9e2cc56e3a82c885155dd3b257935c9b46ba16b","observation_id":"f389ac12-1c55-47f5-80ad-99753d11967f","resolution":{"observed_at":"2026-08-12T13:46:02.532602Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.516037Z","title":"Static block floating-point quantization for convolutional neural networks on fpga,","venue":null,"work_id":"10a32339-d6a8-4224-9fb4-ef246690af60","year":2019},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.012376Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:218cd6b381771c265ac724d510a627acac8d4b9455fbec4e39412dea027d876b","observation_id":"01002d37-24f7-4610-8d4a-c2d9c229c290","resolution":{"observed_at":"2026-08-12T13:46:02.520801Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.502032Z","title":"Optq: Accurate quantization for generative pre-trained transformers,","venue":null,"work_id":"df79867d-09c4-49cd-a058-aa2d6cdb1165","year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.016683Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:4a5b52bb8773dd470f3ef53b37efb8573419dbe779ac3d91d99f4bd71a6247ae","observation_id":"553aa229-a25a-4769-a67b-c312da294cc8","resolution":{"observed_at":"2026-08-12T13:46:02.506326Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.06001","last_updated":"2024-10-09T06:09:41Z","snapshot_observed_at":"2026-08-14T05:37:16.725999Z","submitted_at":"2024-05-09T11:49:05Z","title":"LLMC: Benchmarking Large Language Model Quantization with a Versatile Compression Toolkit","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.06001","snapshot_observed_at":"2026-08-12T13:46:01.020686Z","title":"Llmc: Benchmarking large language model quantization with a versatile compression toolkit,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.020686Z"},"links":{"cited_paper":"/paper/2405.06001","citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:51510eab31169c1bb008e6ea6b74f3634803d254b7406414be3ea2208ef196ac","observation_id":"be801d61-b0fc-4bb3-a27d-e66b430b3b5c","resolution":{"observed_at":"2026-08-12T13:46:01.020686Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.482663Z","title":"Boost: block minifloat-based on-device cnn training accelerator with transfer learning,","venue":null,"work_id":"c56422d3-c614-43f0-8d96-5a9d903f0455","year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.025336Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:f3eb6ab3dd2b527e5c80b7ff97bc4abbf169ffb1dc4547cd7ad32807fe778778","observation_id":"6fcd9706-878e-4c1c-a67b-2c44d6dc4687","resolution":{"observed_at":"2026-08-12T13:46:02.488289Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.468136Z","title":"Olive: Accelerating large language models via hardware- friendly outlier-victim pair quantization,","venue":null,"work_id":"995505cf-b142-4277-82ed-882a51d2b1f1","year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.029221Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:59800c6dbd20a7a8e15610889b2c5a2998302b67b6a4a183a27666f7b3dcc2a9","observation_id":"2efccf6d-a519-4127-b9b4-af72090c2eac","resolution":{"observed_at":"2026-08-12T13:46:02.472758Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.454996Z","title":"Ese: Efficient speech recognition engine with sparse lstm on fpga,","venue":null,"work_id":"283d9bd6-e0f7-438b-b0dd-101efb8b33d0","year":2017},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.033280Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:586899f85f3a32afbe9827040fcc7d88349f88e93f235a590d7f4cb12a882c8b","observation_id":"734655d4-95bf-4ea4-b953-1923a5d8a9fe","resolution":{"observed_at":"2026-08-12T13:46:02.459389Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.18079","last_updated":"2025-05-28T18:58:29Z","snapshot_observed_at":"2026-08-13T04:29:50.330115Z","submitted_at":"2024-01-31T18:58:14Z","title":"KVQuant: Towards 10 Million Context Length LLM Inference with KV Cache Quantization","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.18079","snapshot_observed_at":"2026-08-12T13:46:01.037498Z","title":"Kvquant: Towards 10 million context length llm inference with kv cache quantization,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.037498Z"},"links":{"cited_paper":"/paper/2401.18079","citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:534a8b48db9b47dd61b37a8a64003477f860679679d3f6077adbb1d5c325b126","observation_id":"e31e43a2-f16c-464b-a096-57ffd31fb592","resolution":{"observed_at":"2026-08-12T13:46:01.037498Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.440271Z","title":"A precision-scalable risc-v dnn processor with on-device learning capability at the extreme edge,","venue":null,"work_id":"bf578dd9-306c-455b-a08f-fd145e3e1725","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.042306Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:8f4dcd0bb7da8cfff3e8bb094bf8e82ae153aa5ac509c97d999cc5d6ce844bed","observation_id":"acb4a88b-71fd-425a-a4d4-7557df0a3b61","resolution":{"observed_at":"2026-08-12T13:46:02.445583Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.424203Z","title":"Mind the gap: Attainable data movement and operational intensity bounds for tensor algorithms,","venue":null,"work_id":"1f9f30dd-abd6-4332-8674-edea1ea0e596","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.046677Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:9287aa6d79529e4f8706ac1d22d5efdecc48cc9ea37e2eaefb52302d69058bdb","observation_id":"3c4a177a-2705-4842-99c8-b7d2c8a2e8a7","resolution":{"observed_at":"2026-08-12T13:46:02.429805Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.404127Z","title":"Figna: Integer unit-based accel- erator design for fp-int gemm preserving numerical accuracy,","venue":null,"work_id":"58a226bb-c343-462d-b873-7abdec494ad9","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.050530Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:c51b5e4ff2f98bbe689f47f7d8c207243500ba3727ebc8ea9c02bcf67387cda3","observation_id":"9feedc6f-68ee-45ee-b614-0cd0da64d635","resolution":{"observed_at":"2026-08-12T13:46:02.410522Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.387145Z","title":"Perplexity—a measure of the difficulty of speech recognition tasks,","venue":null,"work_id":"0e8e5695-e332-454c-b565-ce92e62129b3","year":1977},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.054333Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:fcf3dd7a7bf29241d9d0c30f8d358fd0f1357403f8801c11516b04a5ae2fe6bf","observation_id":"b971c7e3-66b4-44ab-9994-26903bdfdec2","resolution":{"observed_at":"2026-08-12T13:46:02.392252Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.366147Z","title":"Mr. biq: Post-training non- uniform quantization based on minimizing the reconstruction error,","venue":null,"work_id":"1524a179-17f1-4406-b2e5-d34985500d9f","year":2022},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.060235Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:053a96fab276af1fea58bc9571099e058a6bd839d86485ec6615afc9804fe46a","observation_id":"cca7ac48-4bed-454b-91a7-e725e24eecce","resolution":{"observed_at":"2026-08-12T13:46:02.371118Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.351155Z","title":"Biqgemm: matrix multiplication with lookup table for binary-coding-based quan- tized dnns,","venue":null,"work_id":"1df7789d-880b-4ae2-90b7-3a19dfc9d87f","year":2020},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.064416Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:34e1171c8e70189080f49c4ec75722400a43fc0c05b2e1ca66d0045011dc6f7a","observation_id":"5a868e5d-84ae-43b1-a764-91759d50d4b7","resolution":{"observed_at":"2026-08-12T13:46:02.356272Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.336265Z","title":"Ten lessons from three generations shaped google’s tpuv4i: Industrial product,","venue":null,"work_id":"c9012d61-ddeb-4868-b13f-f0f1de683929","year":2021},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.069776Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:6bbc3575a0107c225dc99d05313b2d1eb16a528e6efffc737d7368ef3104f557","observation_id":"28571475-c8a3-430d-97f8-fb952857a7fa","resolution":{"observed_at":"2026-08-12T13:46:02.341385Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.321905Z","title":"Stripes: Bit-serial deep neural network computing,","venue":null,"work_id":"669eb79d-6cce-4f5c-9f7d-d91f6fd10eb2","year":2016},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.073787Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:736f19d6cee1b9074a6d42e27d04a2bf744c32cf850840903cb20218c7da97ff","observation_id":"fe860abe-bc83-47e5-b316-3a520278d183","resolution":{"observed_at":"2026-08-12T13:46:02.326657Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.078272Z","title":"A survey of gpt-3 family large language models including chatgpt and gpt-4,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.078272Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:058f1d3a2e981f45c6bb565b5e93e66dcfef9f5a63751436d76aa7af954d30a0","observation_id":"94942890-0859-4f55-b5a3-4a08a3024924","resolution":{"observed_at":"2026-08-12T13:46:01.078272Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.299167Z","title":"A 95.6-tops/w deep learning inference accelerator with per-vector scaled 4-bit quantization in 5 nm,","venue":null,"work_id":"50f0bbdd-1a2b-416b-9f36-d1434307ca54","year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.082459Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:e7f9057aacd11632ada0a7d728bd7d486725e29456726f64f367288550904b2e","observation_id":"02267acb-faf9-41a0-8f47-18c580f1c56c","resolution":{"observed_at":"2026-08-12T13:46:02.304176Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.284082Z","title":"Compressed context mem- ory for online language model interaction,","venue":null,"work_id":"fc5b7f35-fd77-4b8e-b9b8-30f65d424d54","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.086446Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:deb3ed03e5de208123dbeccd883527a4d12d9b3fa6bc6c8d4c2eb17c9cfb8e1e","observation_id":"e01546cb-3e38-4a57-a7db-f2803705ecdf","resolution":{"observed_at":"2026-08-12T13:46:02.288380Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.269127Z","title":"Dacapo: Accelerating continuous learning in autonomous systems for video analytics,","venue":null,"work_id":"cb2ecaff-e4a4-4ea7-b200-e56f1be30970","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.090475Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:8a44b195a02197ae4aed454994cdc2c01ad79cc18d50cbe66464121c1f1afb1e","observation_id":"ba442fe0-bdc1-4a37-be6d-7f271a5796e7","resolution":{"observed_at":"2026-08-12T13:46:02.273986Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.255303Z","title":"Winning both the accuracy of floating point activation and the simplicity of integer arithmetic,","venue":null,"work_id":"259f548d-6dd6-42f1-8037-0df7e9a2a870","year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.096261Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:20aa79c3a0eaf249c619b6faee2d7a2ec8d60e0e6851bcd923aa23510307d48b","observation_id":"048c2a80-8910-4127-b228-ffba59500712","resolution":{"observed_at":"2026-08-12T13:46:02.260399Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.239899Z","title":"One-shot model for mixed-precision quantization,","venue":null,"work_id":"b21a974c-0107-43ed-bf83-303c567656c8","year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.101397Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:fa6dae56e1441c4833f01eb38a1ffa23ca299bc5bb9962fa7f77674e5e1e39ba","observation_id":"05f82623-88e2-48b1-9229-de66136b466e","resolution":{"observed_at":"2026-08-12T13:46:02.244419Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.226728Z","title":"Flexpoint: An adaptive numerical format for efficient training of deep neural networks,","venue":null,"work_id":"3f85079d-3bd7-4a04-a969-7d14af7c09f6","year":2017},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.106244Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:1767355acad5dbcd0aa19b0cc022dffff77b5b16b099cf3bfb823dd53a2f6baa","observation_id":"6de4f875-f11c-4ed3-9f6a-a87b72e63cee","resolution":{"observed_at":"2026-08-12T13:46:02.231230Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.214070Z","title":"Tender: Accelerating large language models via tensor decomposition and runtime requantization,","venue":null,"work_id":"8e02cae2-0283-4d73-a37d-07a663ad8bba","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.110394Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:baacf01a92c740e718e47e3bcc9b5cbe2542d4da1ab3b416486e3bc01b0e7e7a","observation_id":"6ff985f6-a23d-486b-8f4d-8d73abf68337","resolution":{"observed_at":"2026-08-12T13:46:02.218324Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.201351Z","title":"Bitcluster: Fine-grained weight quantization for load-balanced bit-serial neural network accelerators,","venue":null,"work_id":"b60c5c10-cea7-4e39-a0d1-a85594a42862","year":2022},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.114610Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:a90dd2b61c52bf8513d9e5bb30fe45fa9aa8b94f9bcc081d0b85ebe32d18644a","observation_id":"9c0e6188-8cdf-4159-85a6-5069701f9205","resolution":{"observed_at":"2026-08-12T13:46:02.206101Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.185539Z","title":"Norm tweaking: High-performance low-bit quantization of large language models,","venue":null,"work_id":"c1e22cd5-601b-4f41-8da3-d1d1f0a58d54","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.118186Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:83083e994529aaa8b3752984cfb4688d8d546a08fab09b839d6248021d0acf69","observation_id":"19bd6321-1c54-4f04-8dc9-116e8de5dff2","resolution":{"observed_at":"2026-08-12T13:46:02.190206Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.171845Z","title":"Geo: Generation and execution optimized stochastic computing accelerator for neural networks,","venue":null,"work_id":"72823668-296a-466a-8cbb-4fbd58c2758a","year":2021},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.122265Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:90b334f1e3cddfb967204f366996b99931074d020ad6330de55fcbdb0ab226e8","observation_id":"e3b6e19c-f006-4094-a4f2-de465d4cb624","resolution":{"observed_at":"2026-08-12T13:46:02.177003Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.154514Z","title":"Quasar-vit: Hardware-oriented quantization-aware architecture search for vision transformers,","venue":null,"work_id":"53e0675e-a2df-470a-b920-99d83809e37a","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.126214Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:60b708a87f1a35d36069b2bed41d8af74cb5ebb55828162beb953cfe43d578c8","observation_id":"fd869e77-bcc5-47ed-ae71-1f27620566bb","resolution":{"observed_at":"2026-08-12T13:46:02.160011Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.139309Z","title":"High-performance fpga-based cnn accelerator with block-floating-point arithmetic,","venue":null,"work_id":"36111061-6492-4b60-9795-6d91e9b96060","year":2019},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.130338Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:26f27e99d04ca4ec4bddaa74aa094bd1311c62e9fa3372fd5e2df0abc9557e36","observation_id":"995b9f46-e3c0-4229-aeed-91e1b6697a4a","resolution":{"observed_at":"2026-08-12T13:46:02.144687Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.121523Z","title":"Awq: Activation-aware weight quan- tization for llm compression and acceleration,","venue":null,"work_id":"6c99f929-c8f6-4643-ad38-4e550530ce0b","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.134742Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:acd88637126dbf48b1875034755a776897ddeb86fe03e58c6f9bc6bedc82c97b","observation_id":"ac37e816-1469-47bf-b6f8-e6b369beb443","resolution":{"observed_at":"2026-08-12T13:46:02.127154Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.04532","last_updated":"2025-05-01T02:14:05Z","snapshot_observed_at":"2026-08-13T00:12:52.272658Z","submitted_at":"2024-05-07T17:59:30Z","title":"QServe: W4A8KV4 Quantization and System Co-design for Efficient LLM Serving","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.04532","snapshot_observed_at":"2026-08-12T13:46:01.139346Z","title":"Qserve: W4a8kv4 quantization and system co-design for efficient llm serving,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.139346Z"},"links":{"cited_paper":"/paper/2405.04532","citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:cd8a62a9fc66e05a51171a23f5e161040cd8c45e37aeb8478426ffabb7aaf759","observation_id":"0bd651d8-2194-4958-bb11-36374a362fd2","resolution":{"observed_at":"2026-08-12T13:46:01.139346Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.17888","last_updated":"2023-05-29T05:22:11Z","snapshot_observed_at":"2026-08-13T11:30:33.812255Z","submitted_at":"2023-05-29T05:22:11Z","title":"LLM-QAT: Data-Free Quantization Aware Training for Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.17888","snapshot_observed_at":"2026-08-12T13:46:01.145166Z","title":"Llm-qat: Data-free quantization aware training for large language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.145166Z"},"links":{"cited_paper":"/paper/2305.17888","citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:566b560be9ecd013d4191f17bdcb9f6cf903e12a082f39854c9e046414a84725","observation_id":"96581e0b-fb9d-4487-9eeb-24b973f4971b","resolution":{"observed_at":"2026-08-12T13:46:01.145166Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.106782Z","title":"Kivi: A tuning-free asymmetric 2bit quantization for kv cache,","venue":null,"work_id":"a9e1dfd3-7694-48d8-bc7c-7fc607d28cfb","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.149988Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:d269556bbe5928f8d3c6d703613dddb09eed9158eb3c7ba919dbd16b06ae3232","observation_id":"938449d2-5a6d-4ae6-93af-c87b1417f81a","resolution":{"observed_at":"2026-08-12T13:46:02.111294Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.089499Z","title":"Dis- tilling bit-level sparsity parallelism for general purpose deep learning acceleration,","venue":null,"work_id":"cb5ce676-8954-4496-a93f-66c0b58a50df","year":2021},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.154781Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:6e3f27461ea4b8655c01528aa76a1d14a64b4a9a4bb9810c153b41b10ffd982e","observation_id":"9fac23e3-43b4-470b-9fe0-a571e64a12c3","resolution":{"observed_at":"2026-08-12T13:46:02.094997Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.060174Z","title":"Keep the cost down: A review on methods to optimize llm’s kv-cache consumption,","venue":null,"work_id":"1436f428-8a20-4a8b-917a-fec16845daf5","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.159281Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:756e9b8b5c8255da98f90e2314d3e02bb126ae3aefbfaf06e43f99f434fb4c27","observation_id":"9f43329e-2dac-445d-80dc-259f8ef78496","resolution":{"observed_at":"2026-08-12T13:46:02.071054Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.17870","last_updated":"2024-10-18T02:01:18Z","snapshot_observed_at":"2026-08-13T19:08:37.399751Z","submitted_at":"2024-09-26T14:17:58Z","title":"Efficient Arbitrary Precision Acceleration for Large Language Models on GPU Tensor Cores","version":2},"cited_work":{"arxiv_id":"2409.17870","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.17870","snapshot_observed_at":"2026-08-12T13:46:01.503179Z","title":"Efficient Arbitrary Precision Acceleration for Large Language Models on GPU Tensor Cores","venue":"cs.LG","work_id":"e493c2d3-c1a8-4bc8-b65e-e65a24f82326","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.164716Z"},"links":{"cited_paper":"/paper/2409.17870","citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:9558a0afe4189e573d4f0c1a2f04aa6a20a5dca398bee0f8b1c5232d0a3eef5f","observation_id":"fda3f6a9-e9e4-4d15-b6ee-fed49585f9ee","resolution":{"observed_at":"2026-08-12T13:46:01.510626Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.040470Z","title":"Fpnew: An open-source multiformat floating-point unit architecture for energy-proportional transprecision computing,","venue":null,"work_id":"3e92e701-805e-4134-87df-809bdc38887c","year":2020},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.169746Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:9fee3cc08eef033793dc1d6c3b84f68eebe794d4cc9c4f5ee049568ee02d9c0a","observation_id":"92cb1726-fe3b-4af0-9975-a35027aa97d6","resolution":{"observed_at":"2026-08-12T13:46:02.047111Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.023923Z","title":"The penn treebank: Anno- tating predicate argument structure,","venue":null,"work_id":"00e995ad-832f-4056-a84c-9946c25afdd8","year":1994},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.175099Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:e23c21a14d2fcb98ba164bc553a7b7039bf5f6e6bf932187554dce88a61c84ec","observation_id":"ab2800b6-3577-4014-99cc-55e9691b7a41","resolution":{"observed_at":"2026-08-12T13:46:02.029380Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.005235Z","title":"Pointer sentinel mix- ture models,","venue":null,"work_id":"062560e2-1a3f-42fa-94bf-cba761eed757","year":2017},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.180120Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:df66bbed99878d04c4897d4c865fd6c42f9437f99e5d4923c725eca2c81d9ee3","observation_id":"53655596-273c-437f-8332-4a6e8b23254c","resolution":{"observed_at":"2026-08-12T13:46:02.011649Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.989325Z","title":"Flexblock: A flexible dnn training accelerator with multi-mode block floating point support,","venue":null,"work_id":"a6c16ab3-feeb-467e-add5-c97e14fc513c","year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.184647Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:6eaada179fda84de963c17f0851dee52c1cc31672bb201144abdff239b5403e5","observation_id":"c70b0fc3-7afe-4fca-9688-f206d80ec28d","resolution":{"observed_at":"2026-08-12T13:46:01.994256Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.970911Z","title":"Cutlass,","venue":null,"work_id":"17a07bd0-aa4c-4f69-8b9c-e0c1c2504015","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.189062Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:9c552a299c8dbc21990ba02949b266487419df9da6604d663d9bea5bcaa972a9","observation_id":"42f71867-f06c-4e26-8f0f-df6241d81ce7","resolution":{"observed_at":"2026-08-12T13:46:01.977580Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-12T13:46:01.193821Z","title":"Gpt- 4 technical report,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.193821Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:5c4740b540eb676587edd0cb9497ceb7602cddf1ecf1386b2c40eb9dcfa7cc8c","observation_id":"aaa2dfc7-d314-412d-b7be-4e114a599e64","resolution":{"observed_at":"2026-08-12T13:46:01.193821Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.950306Z","title":"LUT-GEMM: Quantized matrix multipli- cation based on LUTs for efficient inference in large-scale generative language models,","venue":null,"work_id":"c6d48234-d0ff-4f62-bbd5-c970cb0c682a","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.198965Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:ba801664c9f5fbd5d039be3b26fddd501d3a72cede4aa523c3d62def33b550ac","observation_id":"38f48fab-2824-488e-99e6-22a062854016","resolution":{"observed_at":"2026-08-12T13:46:01.956504Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.930541Z","title":"Exploring the limits of transfer learning with a unified text-to-text transformer,","venue":null,"work_id":"11737760-2828-4127-b5d9-ea7fdf01907d","year":2020},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.203051Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:ebfeb040dff2a8e586ecafe7af0e2340cd4ee4ccd33d711eb711a30c374f05f7","observation_id":"842a34fe-7a8a-445c-af99-adb9fc1a11ab","resolution":{"observed_at":"2026-08-12T13:46:01.937328Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.913979Z","title":"Omniquant: Omnidirectionally calibrated quantization for large language models,","venue":null,"work_id":"f29737a1-cf8b-4dfa-ab17-d89d589a8ec8","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.207048Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:ef0bcbb23f9d3d83a041164853ff04028b17ad60d12c70efbc5df871fb6d9cbc","observation_id":"a2e29963-82d3-4df9-bd66-09724eabba27","resolution":{"observed_at":"2026-08-12T13:46:01.918778Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.896755Z","title":"Bit fusion: Bit-level dynamically composable architecture for accelerating deep neural network,","venue":null,"work_id":"c09bd469-c3ea-4c65-a58a-02413c431444","year":2018},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.211499Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:e59e68046779cc2e26b5d4337d89517423bfebf6253894cc2f7403dbd9e4492c","observation_id":"0c831a98-4474-4151-be7d-780a15bb36ad","resolution":{"observed_at":"2026-08-12T13:46:01.902106Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.877610Z","title":"Bitwave: Exploiting column-based bit-level sparsity for deep learning accelera- tion,","venue":null,"work_id":"bae62f10-11a2-41cd-bb48-8b06b64d5fef","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.215760Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:5bd0fdd2635a1259a3763b8de25c171772cb4609173b6bcb7d080a14578d90b5","observation_id":"6fc9dfc1-34c9-4723-bbb3-a58c01237086","resolution":{"observed_at":"2026-08-12T13:46:01.883768Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.861263Z","title":"Dissecting tensor cores via microbenchmarks: Latency, throughput and numeric behaviors,","venue":null,"work_id":"39c31d5f-f5ae-4dfb-86ae-b3105c5a9a38","year":2022},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.219759Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:16389608235adbccdbdfbb3540593c5bc3e05ddbf60ff8a8ea854f155c8d0c65","observation_id":"29c5eea0-0148-490a-bccc-b7a1fa42f4ca","resolution":{"observed_at":"2026-08-12T13:46:01.866350Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.08295","last_updated":"2024-04-16T12:52:47Z","snapshot_observed_at":"2026-08-03T03:29:01.959523Z","submitted_at":"2024-03-13T06:59:16Z","title":"Gemma: Open Models Based on Gemini Research and Technology","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.08295","snapshot_observed_at":"2026-08-12T13:46:01.224872Z","title":"Gemma: Open models based on gemini research and technology,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.224872Z"},"links":{"cited_paper":"/paper/2403.08295","citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:56fbd4e1000a9c0276438b73ff520e579954011f0bd1f352456f203adec645cd","observation_id":"9f38003d-2c2e-46f5-a22d-73d7d17f5816","resolution":{"observed_at":"2026-08-12T13:46:01.224872Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.842387Z","title":"Bebert: Efficient and robust binary ensemble bert,","venue":null,"work_id":"c597c36f-1992-4d40-b9d9-da5617f2ec8c","year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.230172Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:90620109b6ed5fc1ae67985ef4ca345a57d1058cac46287f3d1257a257abeeff","observation_id":"141d78a5-df15-482d-b957-e9c47fd67d6a","resolution":{"observed_at":"2026-08-12T13:46:01.847626Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-12T13:46:01.235170Z","title":"Llama: Open and efficient foundation language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.235170Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:0a3a74e206bd7d0272b6120238db679d3771f75ab5b4727a262feac37b019890","observation_id":"11d12725-bcca-489b-a2a9-cfa99513dfc3","resolution":{"observed_at":"2026-08-12T13:46:01.235170Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-12T13:46:01.241211Z","title":"Llama 2: Open foundation and fine-tuned chat models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.241211Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:15d74d7fda24da88e2d859c74fe0cd5b358a8cdd97bd9ecec704c5e372d24da1","observation_id":"5c8e640e-b043-4082-8752-c673a4b40678","resolution":{"observed_at":"2026-08-12T13:46:01.241211Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04396","last_updated":"2024-06-04T04:51:52Z","snapshot_observed_at":"2026-08-13T04:24:47.509375Z","submitted_at":"2024-02-06T20:52:12Z","title":"QuIP#: Even Better LLM Quantization with Hadamard Incoherence and Lattice Codebooks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.04396","snapshot_observed_at":"2026-08-12T13:46:01.246476Z","title":"Quip#: Even better llm quantization with hadamard incoherence and lattice codebooks,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.246476Z"},"links":{"cited_paper":"/paper/2402.04396","citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:cc2f7329f63debc833000b0c049ab8f2e9f54ac20b909518b07fe18e867f1be0","observation_id":"46eab912-d74a-47a8-b1b4-90ff15b5743c","resolution":{"observed_at":"2026-08-12T13:46:01.246476Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.824674Z","title":"Bsvit: A bit-serial vision transformer accelerator exploiting dynamic patch and weight bit-group quantization,","venue":null,"work_id":"875a4acd-0c89-418c-9aad-2f89d93ee3c3","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.252804Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:71c9dd6367340efdaec21b8b615dd7a5ffc0806c2462d8bfaa8600812644356d","observation_id":"8714c3f3-58f9-4e62-b876-960241b8ce18","resolution":{"observed_at":"2026-08-12T13:46:01.830986Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.808713Z","title":"Haq: Hardware-aware automated quantization with mixed precision,","venue":null,"work_id":"50dfaac0-e71d-4482-9178-83f6908e95c9","year":2019},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.257602Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:19f3fe56909c7b8bd500caa89cee84770ca255a0656369d8705f201c9d1b4039","observation_id":"7117e91a-1dc4-4735-8b60-08834df4a5c0","resolution":{"observed_at":"2026-08-12T13:46:01.813986Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.790812Z","title":"Outlier suppression: Pushing the limit of low-bit transformer language models,","venue":null,"work_id":"fe61d7a9-23c5-4458-b6b7-f032a542d2ed","year":2022},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.263418Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:e29de4ee7d6cf28ffe8813959140839c9cc95abd16dbf6f0bbf7b1d1ae090fa9","observation_id":"7357fcfa-01dc-47ce-984c-24f3c5b8c6d7","resolution":{"observed_at":"2026-08-12T13:46:01.796440Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.774302Z","title":"Quant-llm: Accelerating the serving of large language models via fp6- centric algorithm-system co-design on modern gpus,","venue":null,"work_id":"4829e6c8-09a7-47b1-b8b4-c85af7d960de","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.268642Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:8628f662040b28f82da480103937c30696a14b14a18fa08a147d7371747523b6","observation_id":"e45a072e-fb67-41d6-b90b-9c748cf3f417","resolution":{"observed_at":"2026-08-12T13:46:01.780173Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.757965Z","title":"Smoothquant: Accurate and efficient post-training quantization for large language models,","venue":null,"work_id":"3081fba4-1f44-48bb-9f73-cc65dc4cb8ef","year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.274153Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:9eb4f0ed4a9606823625778659bd7c306d2911b44144e38ed6564bfcfae39507","observation_id":"a71eaf87-6323-4f73-9d05-19aca2ce0d6b","resolution":{"observed_at":"2026-08-12T13:46:01.762681Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.739649Z","title":"Efficient streaming language models with attention sinks,","venue":null,"work_id":"6aa653da-9fc5-427f-8071-87b76c356662","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.279979Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:8e7762046077d391acbd307b94ec436cab7fb5ebd0651cddd2f7a854d160ca78","observation_id":"bcbc3551-c787-4ab8-a8e0-5a7133c1a315","resolution":{"observed_at":"2026-08-12T13:46:01.745057Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.11295","last_updated":"2024-11-29T11:47:55Z","snapshot_observed_at":"2026-08-13T04:17:01.673281Z","submitted_at":"2024-02-17T14:26:57Z","title":"OneBit: Towards Extremely Low-bit Large Language Models","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.11295","snapshot_observed_at":"2026-08-12T13:46:01.284612Z","title":"Onebit: Towards extremely low-bit large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.284612Z"},"links":{"cited_paper":"/paper/2402.11295","citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:b61df1d6d2fdc08ae7fb621bc83c2ca133bf8275cf5bc767b71a6dcb3774d417","observation_id":"40732215-a09e-41ab-9da8-fce2c45ab70d","resolution":{"observed_at":"2026-08-12T13:46:01.284612Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.719478Z","title":"Kv cache compression, but what must we give in return? a comprehensive benchmark of long context capable approaches,","venue":null,"work_id":"65b30caa-15f5-4833-bb89-9bb89e6af20a","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.289754Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:02ea8040ef487a89012e8eacd4505973f8e87b4aec381cea0f74a3e29ff20d64","observation_id":"486b959b-41fa-45b4-854b-7615f86f023f","resolution":{"observed_at":"2026-08-12T13:46:01.726050Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.16363","last_updated":"2024-05-01T20:42:28Z","snapshot_observed_at":"2026-08-13T04:10:18.599255Z","submitted_at":"2024-02-26T07:33:05Z","title":"LLM Inference Unveiled: Survey and Roofline Model Insights","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.16363","snapshot_observed_at":"2026-08-12T13:46:01.294427Z","title":"Llm inference unveiled: Survey and roofline model insights,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.294427Z"},"links":{"cited_paper":"/paper/2402.16363","citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:4ae17df42d5169a83e1064ce86e1c069b62fd64c930797ce7bb41a8b117f4132","observation_id":"ff4d496e-4129-4775-9c74-ebbb851dd57d","resolution":{"observed_at":"2026-08-12T13:46:01.294427Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.697876Z","title":"Mokey: Enabling narrow fixed-point inference for out-of-the-box floating-point transformer models,","venue":null,"work_id":"9aaf271f-e395-4b17-8d5e-bcaf4867339e","year":2022},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.299569Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:7f659726a6d3f0c0ea53749feae24407f18ca11b8ea549760d7568ed9035dd29","observation_id":"85aeb8e5-3c60-483a-87ce-7b2378c2c551","resolution":{"observed_at":"2026-08-12T13:46:01.705368Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.676573Z","title":"Fast: Dnn training under variable precision block floating point with stochastic rounding,","venue":null,"work_id":"75fc523d-e5d1-4bd3-a05b-d51abfba1272","year":2022},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.303655Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:49661d797645cf9e1619d7a611fc1de9cb32f2a8beb69eefb9ec31e2e8396a6a","observation_id":"aca341d1-8958-4999-ade4-1e25ceec5f95","resolution":{"observed_at":"2026-08-12T13:46:01.682402Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.01068","last_updated":"2022-06-21T17:04:40Z","snapshot_observed_at":"2026-08-06T03:13:37.403059Z","submitted_at":"2022-05-02T17:49:50Z","title":"OPT: Open Pre-trained Transformer Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.01068","snapshot_observed_at":"2026-08-12T13:46:01.308601Z","title":"Opt: Open pre-trained transformer language models,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.308601Z"},"links":{"cited_paper":"/paper/2205.01068","citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:e19c6507d3aeb63c377606e54600ddea6f743e1aad34be7ed031782acde7c470","observation_id":"50f869ac-ab0d-42bf-85f6-51a7c069feca","resolution":{"observed_at":"2026-08-12T13:46:01.308601Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.653262Z","title":"Cam: Cache merging for memory-efficient llms inference,","venue":null,"work_id":"b3fdf0ba-d5a1-468a-8cd1-15afec1d0bcd","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.313207Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:118164c9c284acdd852f9fecf158ee10a7c671130b75770a5db434f344a735e7","observation_id":"60706211-3401-4a13-b602-09912f9bbf66","resolution":{"observed_at":"2026-08-12T13:46:01.659492Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.632187Z","title":"H2o: Heavy-hitter oracle for efficient generative inference of large language models,","venue":null,"work_id":"44a41251-74c6-48b7-9f2a-ad6c6699dba7","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.317190Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:3f872aaf9155e941bb8c0bdc4c5d9edda67fe61059b6176d9a9424bacd787da0","observation_id":"a419e5d6-a114-4446-9c43-703baa852076","resolution":{"observed_at":"2026-08-12T13:46:01.639814Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.614579Z","title":"Atom: Low-bit quantization for efficient and accurate llm serving,","venue":null,"work_id":"49b53d44-c5ea-4c3a-9848-6b2ec8757aa0","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.321581Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:f5e2a39e067b6e2ef9527a36ab3cd2c80245ee6106b97c72d042e3b145dd7f5f","observation_id":"5e21a48c-cfbe-473b-a856-68a0bb2292e9","resolution":{"observed_at":"2026-08-12T13:46:01.621468Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","latest_version":1,"primary_category":"cs.AR","snapshot_observed_at":"2026-08-15T19:06:10.368269Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format"},"reference_resolution":{"displayed":89,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":23,"verified_exact":1,"verified_fuzzy":65},"total_outbound_references":89},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"thesis":"As of 16 August 2026, this Paper Citation Record lists 89 of 89 outbound references and 0 inbound Pith citation observations for arXiv:2411.15982."}