{"as_of":"2026-08-15T20:22:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:95d8e09d97d317363a2f27be4a8d3aee44ce6a1517b105edb52bfff213d9530e","coverage":[{"denominator":33,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":33,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T12:38:41.059939Z","state":"measured"},{"denominator":35,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":35,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-30T07:03:08.617257Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-06-30T07:04:20.888331Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"cited_work":{"arxiv_id":"2506.15704","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.15704","snapshot_observed_at":"2026-06-30T07:04:20.888331Z","title":"Learn from the past: Fast sparse indexing for large language model decoding","venue":null,"work_id":"6ab074c0-906d-42d7-80ee-a137f5533fec","year":2025},"citing_paper":{"arxiv_id":"2604.10098","last_updated":"2026-04-11T08:41:33Z","snapshot_observed_at":"2026-08-11T07:40:34.488378Z","submitted_at":"2026-04-11T08:41:33Z","title":"Attention Sink in Transformers: A Survey on Utilization, Interpretation, and Mitigation","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-05-10T16:17:09.834609Z"},"links":{"cited_paper":"/paper/2506.15704","citing_paper":"/paper/2604.10098"},"observation_digest":"sha256:9104c3d9bd30a5dc471e906fefc46795d26262a939ea6793b5e7d456c3a8f6b4","observation_id":"adc68189-911a-4ea2-bb71-b3f964fbeaf9","resolution":{"observed_at":"2026-05-11T09:05:57.988048Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"cited_work":{"arxiv_id":"2506.15704","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.15704","snapshot_observed_at":"2026-06-30T07:04:20.888331Z","title":"Learn from the past: Fast sparse indexing for large language model decoding","venue":null,"work_id":"6ab074c0-906d-42d7-80ee-a137f5533fec","year":2025},"citing_paper":{"arxiv_id":"2606.30389","last_updated":"2026-06-29T14:43:25Z","snapshot_observed_at":"2026-08-13T11:37:17.528311Z","submitted_at":"2026-06-29T14:43:25Z","title":"Predict, Reuse, and Repair: Accelerating Dynamic Sparse Attention for Long-Context LLM Decoding","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-06-30T07:03:08.617257Z"},"links":{"cited_paper":"/paper/2506.15704","citing_paper":"/paper/2606.30389"},"observation_digest":"sha256:a65e08c93a9786a37ea7f656210e6c69983eddfd6b4f6cca24b04d3fe0c0b34c","observation_id":"b692aa6d-3478-4390-a79c-b57fa3cd51c3","resolution":{"observed_at":"2026-06-30T07:04:20.890041Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.15704/citation-record","integrity":"/paper/2506.15704/integrity","json":"/paper/2506.15704/citation-record.json","paper":"/paper/2506.15704"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-07T12:38:33.909659Z","title":"Gpt-4 technical report","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:33.909659Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:cd6cba32f525c52be68ae5af6333ce827d8946e30ee4972301fc6cf19a167377","observation_id":"7e9b751b-e021-4306-9155-4dc11094e21f","resolution":{"observed_at":"2026-08-07T12:38:33.909659Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.14508","last_updated":"2024-06-19T04:00:32Z","snapshot_observed_at":"2026-08-08T03:49:18.086396Z","submitted_at":"2023-08-28T11:53:40Z","title":"LongBench: A Bilingual, Multitask Benchmark for Long Context Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.14508","snapshot_observed_at":"2026-08-07T12:38:34.959797Z","title":"Longbench: A bilingual, multitask benchmark for long context understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:34.959797Z"},"links":{"cited_paper":"/paper/2308.14508","citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:afc14c28b7282321c22522ce1467d3a1533b9774f896bd5b55787d6d0920b204","observation_id":"6905c1fa-97b4-4f8c-8379-e543693552b3","resolution":{"observed_at":"2026-08-07T12:38:34.959797Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:38:42.400875Z","title":"Loki then and now: the trickster against civilization","venue":null,"work_id":"61ca2ab5-6708-4fd3-b20f-dedead07d275","year":2017},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:36.864775Z"},"links":{"citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:8c2233e018a0acc41e52e5de8b747ce3bf7c38a8a6f9ee4f5588d8de8979d8c0","observation_id":"4818a542-f672-4751-9d18-0e618417980e","resolution":{"observed_at":"2026-08-07T12:38:42.432776Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.16179","last_updated":"2024-12-18T17:36:36Z","snapshot_observed_at":"2026-08-14T23:45:06.387992Z","submitted_at":"2024-10-21T16:44:51Z","title":"MagicPIG: LSH Sampling for Efficient LLM Generation","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.16179","snapshot_observed_at":"2026-08-07T12:38:37.033773Z","title":"Magicpig: Lsh sampling for efficient llm generation","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:37.033773Z"},"links":{"cited_paper":"/paper/2410.16179","citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:989124b2ae412e358ba874bbd3e60a3d4bdc043238a5dfb885b54c7e27f2be46","observation_id":"fbc23119-1404-43fb-bcbe-a736885ebc2c","resolution":{"observed_at":"2026-08-07T12:38:37.033773Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2105.03011","last_updated":"2021-05-07T00:12:34Z","snapshot_observed_at":"2026-08-15T18:23:20.413643Z","submitted_at":"2021-05-07T00:12:34Z","title":"A Dataset of Information-Seeking Questions and Answers Anchored in Research Papers","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2105.03011","snapshot_observed_at":"2026-08-07T12:38:37.206520Z","title":"A dataset of information-seeking questions and answers anchored in research papers","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:37.206520Z"},"links":{"cited_paper":"/paper/2105.03011","citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:53442e01c7b3442496db870065494eb84af49e55a8442ffdc4300b02e42c3135","observation_id":"1b64cb96-4555-4e9f-96da-9b6ea86f7653","resolution":{"observed_at":"2026-08-07T12:38:37.206520Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:38:37.366095Z","title":"Human-like episodic memory for infinite context llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:37.366095Z"},"links":{"citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:c8994374ca0b69f64c7578d686b3c331495cb547a7e76837e53996708c2ecdfa","observation_id":"7c429e67-9fbe-4570-8558-ab2d3cd7d6ab","resolution":{"observed_at":"2026-08-07T12:38:37.366095Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.11421","last_updated":"2024-03-18T02:30:23Z","snapshot_observed_at":"2026-08-13T19:01:59.954755Z","submitted_at":"2024-03-18T02:30:23Z","title":"FastDecode: High-Throughput GPU-Efficient LLM Serving using Heterogeneous Pipelines","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.11421","snapshot_observed_at":"2026-08-07T12:38:37.472671Z","title":"Fastdecode: High-throughput gpu-efficient llm serving using heterogeneous pipelines","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:37.472671Z"},"links":{"cited_paper":"/paper/2403.11421","citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:7487f3d6294aab3b0e68295cde9f1a7253694e84165614db1e3540abb135f6e0","observation_id":"a9c97e9d-4fe5-4fba-ba4d-0a63006dc951","resolution":{"observed_at":"2026-08-07T12:38:37.472671Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.06654","last_updated":"2024-08-06T21:48:58Z","snapshot_observed_at":"2026-08-15T18:01:46.669862Z","submitted_at":"2024-04-09T23:41:27Z","title":"RULER: What's the Real Context Size of Your Long-Context Language Models?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.06654","snapshot_observed_at":"2026-08-07T12:38:37.591170Z","title":"Ruler: What’s the real context size of your long-context language models? arXiv preprint arXiv:2404.06654, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:37.591170Z"},"links":{"cited_paper":"/paper/2404.06654","citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:348f5483c5bc4449b20554781784fde419a2ed8bb9698af36648a0c3a97befd5","observation_id":"550258a3-3477-4a84-96ab-1b4fb5775f14","resolution":{"observed_at":"2026-08-07T12:38:37.591170Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.17089","last_updated":"2025-06-04T16:08:50Z","snapshot_observed_at":"2026-08-15T13:15:50.907125Z","submitted_at":"2024-11-26T04:03:14Z","title":"KVPR: Efficient LLM Inference with I/O-Aware KV Cache Partial Recomputation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.17089","snapshot_observed_at":"2026-08-07T12:38:37.745707Z","title":"Efficient llm inference with i/o-aware partial kv cache recomputation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:37.745707Z"},"links":{"cited_paper":"/paper/2411.17089","citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:e958d7e04078b3b73e6b2d6ce20a8544fae8dbc6f6e45ca3ca401240a0541732","observation_id":"8dd36822-23d0-4a69-a797-88e0c5ff4a97","resolution":{"observed_at":"2026-08-07T12:38:37.745707Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.02490","last_updated":"2024-10-30T14:53:22Z","snapshot_observed_at":"2026-08-12T23:30:11.201041Z","submitted_at":"2024-07-02T17:59:56Z","title":"MInference 1.0: Accelerating Pre-filling for Long-Context LLMs via Dynamic Sparse Attention","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.02490","snapshot_observed_at":"2026-08-07T12:38:37.911487Z","title":"Minference 1.0: Accelerating pre-filling for long-context llms via dynamic sparse attention","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:37.911487Z"},"links":{"cited_paper":"/paper/2407.02490","citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:43d3de83e136fa9079c17a4245f597e6be9958000c27c518a3cd7160dfead300","observation_id":"2f11bb7e-adc0-419b-9367-506515bf257a","resolution":{"observed_at":"2026-08-07T12:38:37.911487Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.01142","last_updated":"2024-11-02T05:15:44Z","snapshot_observed_at":"2026-08-12T22:09:19.288246Z","submitted_at":"2024-11-02T05:15:44Z","title":"NEO: Saving GPU Memory Crisis with CPU Offloading for Online LLM Inference","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.01142","snapshot_observed_at":"2026-08-07T12:38:38.015826Z","title":"Neo: Saving gpu memory crisis with cpu offloading for online llm inference","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:38.015826Z"},"links":{"cited_paper":"/paper/2411.01142","citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:a0de9a312d93a6b9ae4cf7d226b509f8fd8a37cc4b0888c262d2b4bf923ca5bd","observation_id":"9e46c474-80d9-4b1b-8ce7-4d9e4401fd52","resolution":{"observed_at":"2026-08-07T12:38:38.015826Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03065","last_updated":"2025-02-20T23:28:01Z","snapshot_observed_at":"2026-08-12T22:31:08.153347Z","submitted_at":"2024-10-04T01:11:09Z","title":"Compute Or Load KV Cache? Why Not Both?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.03065","snapshot_observed_at":"2026-08-07T12:38:38.180039Z","title":"Compute or load kv cache? why not both? arXiv preprint arXiv:2410.03065, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:38.180039Z"},"links":{"cited_paper":"/paper/2410.03065","citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:36d49b467e6605525c7b4237e77ff710025063c4a88854a17768efec13f057ea","observation_id":"a34eb7be-896c-4b89-b6ba-31cd8f00ef2a","resolution":{"observed_at":"2026-08-07T12:38:38.180039Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:38:38.354119Z","title":"Efficient memory management for large language model serving with pagedattention","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:38.354119Z"},"links":{"citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:c22eb4c4489664cfddd5fc6c44d132408eb555f11bbe082b3aa8942f45245c1c","observation_id":"c1bfc8f9-aeab-4816-b3b3-d2d996cf9159","resolution":{"observed_at":"2026-08-07T12:38:38.354119Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:38:38.493837Z","title":"{InfiniGen}: Efficient generative inference of large language models with dynamic {KV} cache management","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:38.493837Z"},"links":{"citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:d6818af58ae3936c801eceeebad9a04ddd05a0e5354dd981d0e65dcaed519bcb","observation_id":"661a023e-a340-4058-832c-8704d5177c46","resolution":{"observed_at":"2026-08-07T12:38:38.493837Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:38:42.138241Z","title":"Snapkv: Llm knows what you are looking for before generation","venue":null,"work_id":"6d8ef366-2897-4d3c-8c27-8c4dcfa75a85","year":2024},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:38.606433Z"},"links":{"citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:ec80a589748315dc6dcbf07b232692fb96a1328f430717d0363e72d1eb133ee6","observation_id":"eada5741-34d4-4ef6-a9c4-bf9fa63b8c1a","resolution":{"observed_at":"2026-08-07T12:38:42.226835Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-08-15T17:27:11.980940Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-07T12:38:38.773218Z","title":"Deepseek-v3 technical report","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:38.773218Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:3a34417c76c61762830a5a920f6e9a798fe5fc0d9c6d4cf2e02c7d458a091664","observation_id":"16d82528-4536-421b-acd9-37cd4177d06c","resolution":{"observed_at":"2026-08-07T12:38:38.773218Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.10516","last_updated":"2024-12-31T07:11:00Z","snapshot_observed_at":"2026-08-15T07:47:03.589816Z","submitted_at":"2024-09-16T17:59:52Z","title":"RetrievalAttention: Accelerating Long-Context LLM Inference via Vector Retrieval","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.10516","snapshot_observed_at":"2026-08-07T12:38:38.885050Z","title":"Retrievalattention: Accelerating long- context llm inference via vector retrieval","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:38.885050Z"},"links":{"cited_paper":"/paper/2409.10516","citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:ece252e55ad51ec6fb5cf042060e290e8c33a3a20d6460fcb59c426c030fda8c","observation_id":"977bba42-95d7-49de-a137-96176e3d0470","resolution":{"observed_at":"2026-08-07T12:38:38.885050Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.03213","last_updated":"2025-06-14T06:17:33Z","snapshot_observed_at":"2026-08-14T21:26:50.903212Z","submitted_at":"2024-12-04T10:58:27Z","title":"ClusterKV: Manipulating LLM KV Cache in Semantic Space for Recallable Compression","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.03213","snapshot_observed_at":"2026-08-07T12:38:39.010317Z","title":"Clusterkv: Manipulating llm kv cache in semantic space for recallable compression","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:39.010317Z"},"links":{"cited_paper":"/paper/2412.03213","citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:0e687f43ce5d62030f60438cdce5b45f8354ce649936dc9795cb8e468792e12c","observation_id":"e878039f-0681-4e7b-9be6-f5d943c4b78e","resolution":{"observed_at":"2026-08-07T12:38:39.010317Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13189","last_updated":"2025-02-18T14:06:05Z","snapshot_observed_at":"2026-08-12T19:30:23.025771Z","submitted_at":"2025-02-18T14:06:05Z","title":"MoBA: Mixture of Block Attention for Long-Context LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13189","snapshot_observed_at":"2026-08-07T12:38:39.114620Z","title":"Moba: Mixture of block attention for long-context llms","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:39.114620Z"},"links":{"cited_paper":"/paper/2502.13189","citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:d2f4bcece8ee42fc4c6b837363f0c8d67fa78e16ef85d414b645c532dcb8672b","observation_id":"bc97e210-8720-4018-962e-cd4d98de5a87","resolution":{"observed_at":"2026-08-07T12:38:39.114620Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04985","last_updated":"2024-09-04T10:04:52Z","snapshot_observed_at":"2026-08-13T05:07:19.369907Z","submitted_at":"2023-12-08T11:47:35Z","title":"SparQ Attention: Bandwidth-Efficient LLM Inference","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04985","snapshot_observed_at":"2026-08-07T12:38:39.247507Z","title":"Sparq attention: Bandwidth-efficient llm inference","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:39.247507Z"},"links":{"cited_paper":"/paper/2312.04985","citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:c42b4715d679acb22037f50fdab9cb4710d4bbb89d7862c4a2e84b8c3de2e59d","observation_id":"1e514596-25c8-486d-96ff-73b94b3258e6","resolution":{"observed_at":"2026-08-07T12:38:39.247507Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12950","last_updated":"2024-01-31T19:47:26Z","snapshot_observed_at":"2026-08-13T04:25:38.282910Z","submitted_at":"2023-08-24T17:39:13Z","title":"Code Llama: Open Foundation Models for Code","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12950","snapshot_observed_at":"2026-08-07T12:38:39.414534Z","title":"Code llama: Open foundation models for code","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:39.414534Z"},"links":{"cited_paper":"/paper/2308.12950","citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:4f0ea813f63150b7cddc845e9d84ca4f2906256b5332c59b4c94c7ebd3b5c2a2","observation_id":"952fef15-d388-47c7-aa55-f96d7c1b1060","resolution":{"observed_at":"2026-08-07T12:38:39.414534Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:38:39.514366Z","title":"Flexgen: High-throughput generative inference of large language models with a single gpu","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:39.514366Z"},"links":{"citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:5ffd94d029e23e645a678ee70c13fc24122444d313c510dbde3343e97008322d","observation_id":"cc38a9bf-8c56-45a5-abae-122c51efc83e","resolution":{"observed_at":"2026-08-07T12:38:39.514366Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21465","last_updated":"2025-04-25T19:40:54Z","snapshot_observed_at":"2026-08-12T22:13:26.137322Z","submitted_at":"2024-10-28T19:08:12Z","title":"ShadowKV: KV Cache in Shadows for High-Throughput Long-Context LLM Inference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21465","snapshot_observed_at":"2026-08-07T12:38:39.632271Z","title":"Shadowkv: Kv cache in shadows for high-throughput long-context llm inference","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:39.632271Z"},"links":{"cited_paper":"/paper/2410.21465","citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:fc01dfbd709d71bd06090691ed29d1692c3df7309b2f5f076be16919b492cccc","observation_id":"9545cbf4-34d8-46f4-9560-9cacafa53065","resolution":{"observed_at":"2026-08-07T12:38:39.632271Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.10774","last_updated":"2024-08-26T21:01:02Z","snapshot_observed_at":"2026-08-14T18:53:06.624928Z","submitted_at":"2024-06-16T01:33:02Z","title":"Quest: Query-Aware Sparsity for Efficient Long-Context LLM Inference","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.10774","snapshot_observed_at":"2026-08-07T12:38:39.786344Z","title":"Quest: Query-aware sparsity for efficient long-context llm inference","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:39.786344Z"},"links":{"cited_paper":"/paper/2406.10774","citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:021cb301c8ce14658d0aafaa947b33874fb689800ffdee5d08fb7a912f601b7e","observation_id":"04f60f49-951a-40a4-a26d-2e8c88f2a8bf","resolution":{"observed_at":"2026-08-07T12:38:39.786344Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-07T12:38:39.932459Z","title":"Llama: Open and efficient foundation language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:39.932459Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:4395f03714c82e57b29a0b66a51ae6b52f1bdece7c0181a8ddc1cd9ca0ec401f","observation_id":"bed9b6f2-dc40-4894-92bb-2aa4d92a1ff7","resolution":{"observed_at":"2026-08-07T12:38:39.932459Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.08454","last_updated":"2024-07-21T02:37:11Z","snapshot_observed_at":"2026-08-12T23:24:15.519470Z","submitted_at":"2024-07-11T12:50:42Z","title":"Model Tells You Where to Merge: Adaptive KV Cache Merging for LLMs on Long-Context Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.08454","snapshot_observed_at":"2026-08-07T12:38:40.064295Z","title":"Model tells you where to merge: Adaptive kv cache merging for llms on long-context tasks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:40.064295Z"},"links":{"cited_paper":"/paper/2407.08454","citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:5fcdb96d2bb8ad48f8c2af404e0156f4e0ad71149e26fac82c7ab17cd6c7b421","observation_id":"6a23d7f7-75e2-45b2-9981-415acf380d00","resolution":{"observed_at":"2026-08-07T12:38:40.064295Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:38:41.844717Z","title":"Infllm: Unveiling the intrinsic capacity of llms for under- standing extremely long sequences with training-free memory.arXiv e-prints, pages arXiv–2402, 2024","venue":null,"work_id":"bea59fa7-0807-4921-a867-0b3b85e1b020","year":2024},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:40.195808Z"},"links":{"citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:d750b568c356b500be585f6dc490599e9f9df0aa1985c28e554d032340ba3cd6","observation_id":"9b3bc751-1820-4855-b643-3b526a4061b9","resolution":{"observed_at":"2026-08-07T12:38:41.973919Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.10819","last_updated":"2024-10-14T17:59:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-14T17:59:58Z","title":"DuoAttention: Efficient Long-Context LLM Inference with Retrieval and Streaming Heads","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.10819","snapshot_observed_at":"2026-08-07T12:38:40.368474Z","title":"Duoattention: Efficient long-context llm inference with retrieval and streaming heads","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:40.368474Z"},"links":{"cited_paper":"/paper/2410.10819","citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:1cc7d7764d67db5f039ad7b0609609712575f59fa9505501b4cdf129fbbda17a","observation_id":"aa115202-2d34-45d5-9d2a-faa4234d69e4","resolution":{"observed_at":"2026-08-07T12:38:40.368474Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17453","last_updated":"2024-04-07T00:56:53Z","snapshot_observed_at":"2026-08-14T06:01:54.549199Z","submitted_at":"2023-09-29T17:59:56Z","title":"Efficient Streaming Language Models with Attention Sinks","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.17453","snapshot_observed_at":"2026-08-07T12:38:40.524261Z","title":"Efficient streaming language models with attention sinks","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:40.524261Z"},"links":{"cited_paper":"/paper/2309.17453","citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:72891d3dec40edb343f0a1dffa63afd2a78263789c18f9f7b182d55605d43acb","observation_id":"d557360f-bcb7-4c09-9807-c01bc5effd5e","resolution":{"observed_at":"2026-08-07T12:38:40.524261Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15383","last_updated":"2025-01-26T03:47:25Z","snapshot_observed_at":"2026-08-11T19:42:36.266059Z","submitted_at":"2025-01-26T03:47:25Z","title":"Qwen2.5-1M Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.15383","snapshot_observed_at":"2026-08-07T12:38:40.663165Z","title":"Qwen2.5-1m technical report","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:40.663165Z"},"links":{"cited_paper":"/paper/2501.15383","citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:98bc034bf752c2a32f6200b9ef6f552da6492a3711a39e5614cda72682797bf7","observation_id":"23115589-32d9-4e42-b217-8997271304a3","resolution":{"observed_at":"2026-08-07T12:38:40.663165Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:38:40.778601Z","title":"Orca: A distributed serving system for {Transformer-Based} generative models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:40.778601Z"},"links":{"citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:fa5991d8cfdd74937f476d6f46dff643f48d8f608d20b81145ccedfb3ab34066","observation_id":"5437d5b9-1181-4359-adaf-152295e111b3","resolution":{"observed_at":"2026-08-07T12:38:40.778601Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.12820","last_updated":"2025-03-30T08:13:50Z","snapshot_observed_at":"2026-08-14T07:48:38.103893Z","submitted_at":"2024-07-01T13:05:42Z","title":"PQCache: Product Quantization-based KVCache for Long Context LLM Inference","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.12820","snapshot_observed_at":"2026-08-07T12:38:40.900312Z","title":"Pqcache: Product quantization-based kvcache for long context llm inference","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:40.900312Z"},"links":{"cited_paper":"/paper/2407.12820","citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:863589cf0e6fd9baeea87788f19e39ae1575d58f952243987e27536071c6146f","observation_id":"0a5774c3-cbdc-4a02-aab5-6c6eaea9326c","resolution":{"observed_at":"2026-08-07T12:38:40.900312Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:38:41.551993Z","title":"H2o: Heavy-hitter oracle for efficient generative inference of large language models","venue":null,"work_id":"d1d5c77d-ec43-42da-b99d-dcba3d9fd8d2","year":2023},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:41.059939Z"},"links":{"citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:f45439ce090122691380e6b7573b9e0209aa2d1193bd6e9e2edd38fa67978bca","observation_id":"0057c0b2-26c2-4e71-9083-60f8c3ffe1db","resolution":{"observed_at":"2026-08-07T12:38:41.662361Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-11T00:38:33.517153Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding"},"reference_resolution":{"displayed":33,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":29,"verified_exact":0,"verified_fuzzy":4},"total_outbound_references":33},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 33 of 33 outbound references and 2 inbound Pith citation observations for arXiv:2506.15704."}