{"as_of":"2026-08-16T02:49:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6058dc507e0186ee31fd4bca019c0fca32aaa38c95236d72db5985e4dbe50f5e","coverage":[{"denominator":39,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":39,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T16:43:44.403049Z","state":"measured"},{"denominator":40,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":40,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-26T08:59:44.970855Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T10:19:47.322402Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"cited_work":{"arxiv_id":"2412.16187","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.16187","snapshot_observed_at":"2026-07-04T10:19:47.322402Z","title":"arXiv:2412.16187 , year =","venue":null,"work_id":"756e04c7-4096-46fb-8dbe-1ce700aed457","year":null},"citing_paper":{"arxiv_id":"2606.22874","last_updated":"2026-06-22T05:39:12Z","snapshot_observed_at":"2026-08-15T01:46:45.403130Z","submitted_at":"2026-06-22T05:39:12Z","title":"SpotAttention: Plug-In Block-Sparse Routing for Pretrained Long-Context Transformers","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-06-26T08:59:44.970855Z"},"links":{"cited_paper":"/paper/2412.16187","citing_paper":"/paper/2606.22874"},"observation_digest":"sha256:ebef7481a6cd4d125dd7260380d8ef6d8d31561af1e5a6a4db8e42ae6936cf89","observation_id":"fe916e22-16d0-46c3-9435-07c511f31da9","resolution":{"observed_at":"2026-07-04T10:19:47.323749Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2412.16187/citation-record","integrity":"/paper/2412.16187/integrity","json":"/paper/2412.16187/citation-record.json","paper":"/paper/2412.16187"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.18223","last_updated":"2026-03-18T05:34:39Z","snapshot_observed_at":"2026-08-14T10:40:26.323157Z","submitted_at":"2023-03-31T17:28:46Z","title":"A Survey of Large Language Models","version":19},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.18223","snapshot_observed_at":"2026-08-11T16:43:43.737452Z","title":"A survey of large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:43.737452Z"},"links":{"cited_paper":"/paper/2303.18223","citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:3c2fc4f6975d09ff2728662825164838fc4e3e940517c965b173f34edc5d0864","observation_id":"e3651cfa-f343-4867-bd9d-09a9e894be82","resolution":{"observed_at":"2026-08-11T16:43:43.737452Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.07682","last_updated":"2022-10-26T05:06:24Z","snapshot_observed_at":"2026-08-02T15:56:35.249569Z","submitted_at":"2022-06-15T17:32:01Z","title":"Emergent Abilities of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.07682","snapshot_observed_at":"2026-08-11T16:43:43.743889Z","title":"Emergent abilities of large language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:43.743889Z"},"links":{"cited_paper":"/paper/2206.07682","citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:74a032d2bbe2611d21b312cc8225e64fedd9641f7489138279de9c5aa9f681c2","observation_id":"e4236455-69ec-4db3-aee9-8a24bd2391c5","resolution":{"observed_at":"2026-08-11T16:43:43.743889Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1409.0473","last_updated":"2016-05-19T21:53:22Z","snapshot_observed_at":"2026-08-12T12:07:33.202888Z","submitted_at":"2014-09-01T16:33:02Z","title":"Neural Machine Translation by Jointly Learning to Align and Translate","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1409.0473","snapshot_observed_at":"2026-08-11T16:43:43.749297Z","title":"Neural machine translation by jointly learning to align and translate","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:43.749297Z"},"links":{"cited_paper":"/paper/1409.0473","citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:23e1495e71ecaea3a424260e5003640fa43cfc99cb87d30d55ca37e6562ef0b3","observation_id":"5ecfb2fd-1cef-467e-a389-84d385e54f09","resolution":{"observed_at":"2026-08-11T16:43:43.749297Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1508.04025","last_updated":"2015-09-20T08:25:52Z","snapshot_observed_at":"2026-08-14T22:36:17.759859Z","submitted_at":"2015-08-17T13:43:19Z","title":"Effective Approaches to Attention-based Neural Machine Translation","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1508.04025","snapshot_observed_at":"2026-08-11T16:43:43.754816Z","title":"Effective approaches to attention-based neural machine translation","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:43.754816Z"},"links":{"cited_paper":"/paper/1508.04025","citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:cf53b3382ef162217c68ddd047bc0e5c963f60e49c88ee5f2658091f4c09a4d5","observation_id":"c6e600da-c089-4186-8d2e-7bc171bac888","resolution":{"observed_at":"2026-08-11T16:43:43.754816Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:43:43.759942Z","title":"Attention is all you need","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:43.759942Z"},"links":{"citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:85b128acb38edd7b3118024262230e92ece40a5957fe0baa3dda9985d85bd1e2","observation_id":"5de0ce26-c5c2-4de4-9d7e-d75b3d7c9e92","resolution":{"observed_at":"2026-08-11T16:43:43.759942Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-11T16:43:43.764731Z","title":"The llama 3 herd of models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:43.764731Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:27a5939b7083239ed8beeabed014ace07c29016561f2a0634154d0791af4eab6","observation_id":"f14c6fb9-6a5b-425e-bd20-9dd21a2ac670","resolution":{"observed_at":"2026-08-11T16:43:43.764731Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.08944","last_updated":"2024-05-14T20:17:22Z","snapshot_observed_at":"2026-08-14T17:28:28.240869Z","submitted_at":"2024-05-14T20:17:22Z","title":"Challenges in Deploying Long-Context Transformers: A Theoretical Peak Performance Analysis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.08944","snapshot_observed_at":"2026-08-11T16:43:43.770122Z","title":"Challenges in deploying long-context transformers: A theoretical peak performance analysis","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:43.770122Z"},"links":{"cited_paper":"/paper/2405.08944","citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:cb05de774e4c4f92f9d915a1d55f521601c182b0c86bd7a6eb45e98960a38895","observation_id":"a3155492-a56a-4612-9f47-99e5983436d1","resolution":{"observed_at":"2026-08-11T16:43:43.770122Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01801","last_updated":"2024-10-29T18:26:09Z","snapshot_observed_at":"2026-08-14T11:45:47.400186Z","submitted_at":"2023-10-03T05:17:08Z","title":"Model Tells You What to Discard: Adaptive KV Cache Compression for LLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01801","snapshot_observed_at":"2026-08-11T16:43:43.775210Z","title":"Model tells you what to discard: Adaptive kv cache compression for llms","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:43.775210Z"},"links":{"cited_paper":"/paper/2310.01801","citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:be54e9c3cd66dcf8db89bc7a0770914a965671036a3f5d47a42eabfb2c5e531c","observation_id":"9f047785-3d64-4c5e-b6cf-04113097f87a","resolution":{"observed_at":"2026-08-11T16:43:43.775210Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:43:45.000486Z","title":"H2o: Heavy-hitter oracle for efficient generative inference of large language models","venue":null,"work_id":"0917f6df-b58e-4c41-8a62-3eea503985f7","year":2024},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:43.780015Z"},"links":{"citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:3f1a6d04292925061db663a848140ea6c1dd450f246ae977ad26c6d22f419ff7","observation_id":"fc3e1cb6-b53e-47b0-b479-de0e621e0bf0","resolution":{"observed_at":"2026-08-11T16:43:45.010530Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:43:43.784590Z","title":"Q-hitter: A better token oracle for efficient llm inference via sparse-quantized kv cache","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:43.784590Z"},"links":{"citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:5dd737d49a55c5841fa2de15301cb4f6c298c2810f6a4b0d376dfeb7aa9eb808","observation_id":"ed8e8fdc-114e-47f7-9766-1afe5c63fb3d","resolution":{"observed_at":"2026-08-11T16:43:43.784590Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11430","last_updated":"2024-11-03T09:42:35Z","snapshot_observed_at":"2026-08-15T06:43:46.557876Z","submitted_at":"2024-06-17T11:35:16Z","title":"A Simple and Effective $L_2$ Norm-Based Strategy for KV Cache Compression","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11430","snapshot_observed_at":"2026-08-11T16:43:43.823950Z","title":"A simple and effective l\\_2 norm-based strategy for kv cache compression","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:43.823950Z"},"links":{"cited_paper":"/paper/2406.11430","citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:808968e86327da2c38dc4fbf49501a5725655dd16d7bdcbff966d40f412b0e45","observation_id":"7fdb926b-1983-4b08-9db8-0cf51bb94c95","resolution":{"observed_at":"2026-08-11T16:43:43.823950Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12335","last_updated":"2024-10-02T00:19:13Z","snapshot_observed_at":"2026-08-15T07:52:10.795203Z","submitted_at":"2024-06-18T07:01:11Z","title":"Attention Score is not All You Need for Token Importance Indicator in KV Cache Reduction: Value Also Matters","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12335","snapshot_observed_at":"2026-08-11T16:43:43.867561Z","title":"Attention score is not all you need for token importance indicator in kv cache reduction: Value also matters","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:43.867561Z"},"links":{"cited_paper":"/paper/2406.12335","citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:a65dca4a0b97154dc3e93ad63454586ce97a4e7dde4100986c9063db09a986dc","observation_id":"6f9099a9-fbee-462b-849f-c7d79f79ca8b","resolution":{"observed_at":"2026-08-11T16:43:43.867561Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:43:44.974895Z","title":"Beyond attentive tokens: Incorporating token importance and diversity for efficient vision transformers","venue":null,"work_id":"998ffa11-8c07-47bf-b19b-175a0754828b","year":2023},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:43.904314Z"},"links":{"citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:a073f9ecfcc06d09240026a1157d28e8cae2702f25f3b01dff09dd2f6e154e4e","observation_id":"62512a98-89f9-406d-8b78-2f6820c8fcfd","resolution":{"observed_at":"2026-08-11T16:43:44.980023Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:43:43.942630Z","title":"Improved approximation algorithms for maximum cut and satisfiability problems using semidefinite programming","venue":null,"work_id":null,"year":1995},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:43.942630Z"},"links":{"citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:f2953d18fcff8b90e521d8ff1b249f5a588db91cccfdb35ce466c21c6fb03b15","observation_id":"1189bb93-5a59-4543-a9cc-6b2b413774fe","resolution":{"observed_at":"2026-08-11T16:43:43.942630Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:43:44.948220Z","title":"Similarity estimation techniques from rounding algorithms","venue":null,"work_id":"08f10802-59cc-43b0-bbfd-6b6c46c09ca4","year":2002},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:44.005871Z"},"links":{"citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:31e209258f2a990ef7ee72beb08a75df082f5858b81d5835cf9cf2b0433085ef","observation_id":"eecaee27-f217-4d3e-beeb-5777a5c3c870","resolution":{"observed_at":"2026-08-11T16:43:44.953190Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-14T02:43:01.480086Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-11T16:43:44.043461Z","title":"Training verifiers to solve math word problems","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:44.043461Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:c057b63d9638ed0d30ac040793a3269fd61f6fe62b9a9704d40da14cb963e45b","observation_id":"efd09993-04b9-448c-ba87-7eaffe876149","resolution":{"observed_at":"2026-08-11T16:43:44.043461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.06654","last_updated":"2024-08-06T21:48:58Z","snapshot_observed_at":"2026-08-15T18:01:46.669862Z","submitted_at":"2024-04-09T23:41:27Z","title":"RULER: What's the Real Context Size of Your Long-Context Language Models?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.06654","snapshot_observed_at":"2026-08-11T16:43:44.096202Z","title":"Ruler: What's the real context size of your long-context language models? arXiv preprint arXiv:2404.06654, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:44.096202Z"},"links":{"cited_paper":"/paper/2404.06654","citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:6fd39ebc695d06c6bf0f3cf0f8aea4a2b40ffb11e62c6bc18a91f5564c75b492","observation_id":"2181495a-8264-4fbe-b2e9-54dbc2db228e","resolution":{"observed_at":"2026-08-11T16:43:44.096202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.14508","last_updated":"2024-06-19T04:00:32Z","snapshot_observed_at":"2026-08-08T03:49:18.086396Z","submitted_at":"2023-08-28T11:53:40Z","title":"LongBench: A Bilingual, Multitask Benchmark for Long Context Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.14508","snapshot_observed_at":"2026-08-11T16:43:44.101225Z","title":"Longbench: A bilingual, multitask benchmark for long context understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:44.101225Z"},"links":{"cited_paper":"/paper/2308.14508","citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:c3bfe8d73185882efbf707c0e3a54fafc4b0b2ba59bc3d70da95fba56dd81dc6","observation_id":"287c502f-ddb5-470d-9d48-50824a674d0b","resolution":{"observed_at":"2026-08-11T16:43:44.101225Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2207.09238","last_updated":"2022-07-19T12:49:02Z","snapshot_observed_at":"2026-08-14T21:45:19.757174Z","submitted_at":"2022-07-19T12:49:02Z","title":"Formal Algorithms for Transformers","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2207.09238","snapshot_observed_at":"2026-08-11T16:43:44.106277Z","title":"Formal algorithms for transformers","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:44.106277Z"},"links":{"cited_paper":"/paper/2207.09238","citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:8f45106e2ce3f2fe51a41f0cccd12eb8a791e6b4f2f909e6cd3d557330c0c4ed","observation_id":"299636b4-cc94-4354-bb92-d4b2ea6fd0cf","resolution":{"observed_at":"2026-08-11T16:43:44.106277Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:43:44.930825Z","title":"Approximate nearest neighbor search in high dimensions","venue":null,"work_id":"d83374a0-4608-4a87-bc3e-9aeb6df07101","year":2018},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:44.111826Z"},"links":{"citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:160f6d7d3fdb1771e35f94563869130401e4af12d2b0dd11b2a93ab93ed94c37","observation_id":"00431976-5634-4013-bab3-96c8d29c84bf","resolution":{"observed_at":"2026-08-11T16:43:44.935968Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:43:44.915807Z","title":"Scissorhands: Exploiting the persistence of importance hypothesis for llm kv cache compression at test time","venue":null,"work_id":"c50f3e0c-e2a1-4b2e-885d-137dde4c2713","year":2024},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:44.118585Z"},"links":{"citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:b8e5628b12a822066d9a6b92d5806e49847796fc8eafe83306a022219381f3f8","observation_id":"0a464099-9a06-4753-acd4-6040b52c7b9a","resolution":{"observed_at":"2026-08-11T16:43:44.920261Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.13245","last_updated":"2023-12-23T17:55:11Z","snapshot_observed_at":"2026-08-15T06:28:09.529747Z","submitted_at":"2023-05-22T17:16:38Z","title":"GQA: Training Generalized Multi-Query Transformer Models from Multi-Head Checkpoints","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.13245","snapshot_observed_at":"2026-08-11T16:43:44.122681Z","title":"Gqa: Training generalized multi-query transformer models from multi-head checkpoints","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:44.122681Z"},"links":{"cited_paper":"/paper/2305.13245","citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:996a4513782967b83a1ba5791dae45b94c06c96f7a66a626d80603a6d65eea2f","observation_id":"73c1e8c1-98c6-4410-a6ba-076d25df1217","resolution":{"observed_at":"2026-08-11T16:43:44.122681Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.08454","last_updated":"2024-07-21T02:37:11Z","snapshot_observed_at":"2026-08-12T23:24:15.519470Z","submitted_at":"2024-07-11T12:50:42Z","title":"Model Tells You Where to Merge: Adaptive KV Cache Merging for LLMs on Long-Context Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.08454","snapshot_observed_at":"2026-08-11T16:43:44.127280Z","title":"Model tells you where to merge: Adaptive kv cache merging for llms on long-context tasks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:44.127280Z"},"links":{"cited_paper":"/paper/2407.08454","citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:01a0457cb2da816c3407965e4603f09e2d34af4030fef3bd9e8b45e1b01d980b","observation_id":"05f3dd23-7762-41ab-9af0-b38f36eee298","resolution":{"observed_at":"2026-08-11T16:43:44.127280Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.14366","last_updated":"2024-09-07T02:52:29Z","snapshot_observed_at":"2026-08-13T00:00:46.126397Z","submitted_at":"2024-05-23T09:43:52Z","title":"MiniCache: KV Cache Compression in Depth Dimension for Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.14366","snapshot_observed_at":"2026-08-11T16:43:44.131384Z","title":"Minicache: Kv cache compression in depth dimension for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:44.131384Z"},"links":{"cited_paper":"/paper/2405.14366","citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:1b39656198f5cd00936689216a45859fe6048ee5acec9c96ac2b46f35a7fbd63","observation_id":"a09a16a1-07cc-4ba0-a737-ae241f0805eb","resolution":{"observed_at":"2026-08-11T16:43:44.131384Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2001.04451","last_updated":"2020-02-18T16:01:18Z","snapshot_observed_at":"2026-07-06T08:50:12.690900Z","submitted_at":"2020-01-13T18:38:28Z","title":"Reformer: The Efficient Transformer","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2001.04451","snapshot_observed_at":"2026-08-11T16:43:44.135397Z","title":"Reformer: The efficient transformer","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:44.135397Z"},"links":{"cited_paper":"/paper/2001.04451","citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:649eb5a3f1cb4f56ffa1a6759f2693c233a516302942244c08b289d9d9892323","observation_id":"13242575-7e09-40ba-91a5-e84f8aaaf053","resolution":{"observed_at":"2026-08-11T16:43:44.135397Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:43:44.901379Z","title":"Kdeformer: Accelerating transformers via kernel density estimation","venue":null,"work_id":"937f3fa4-6079-41bd-91b1-4884aca7e74d","year":2023},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:44.139547Z"},"links":{"citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:03251363157ea53c8aa17906008f2b5046a28e5beb2a2d5a50dfa5ca4fe67081","observation_id":"c9f6a8a4-f159-47b9-8eef-e069afea78ee","resolution":{"observed_at":"2026-08-11T16:43:44.905565Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.05869","last_updated":"2023-12-01T17:43:06Z","snapshot_observed_at":"2026-08-13T05:53:43.374620Z","submitted_at":"2023-10-09T17:05:25Z","title":"HyperAttention: Long-context Attention in Near-Linear Time","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.05869","snapshot_observed_at":"2026-08-11T16:43:44.143968Z","title":"Hyperattention: Long-context attention in near-linear time","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:44.143968Z"},"links":{"cited_paper":"/paper/2310.05869","citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:8f7a1518acdb798dbdcb882c888ccd15bfddde48aff4468f3f48b1a00d495448","observation_id":"f3fb8b69-0aea-43c6-bc60-86eeda4cfb5d","resolution":{"observed_at":"2026-08-11T16:43:44.143968Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.03482","last_updated":"2024-07-18T16:31:29Z","snapshot_observed_at":"2026-08-12T23:49:22.804417Z","submitted_at":"2024-06-05T17:42:05Z","title":"QJL: 1-Bit Quantized JL Transform for KV Cache Quantization with Zero Overhead","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.03482","snapshot_observed_at":"2026-08-11T16:43:44.149261Z","title":"Qjl: 1-bit quantized jl transform for kv cache quantization with zero overhead","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:44.149261Z"},"links":{"cited_paper":"/paper/2406.03482","citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:e0e96257a7be4f700a0a3f08e3254060c6c1da5e6c6eea6d8ba4e571670e325e","observation_id":"4193d3ef-d509-43dd-bfaa-a13c18145727","resolution":{"observed_at":"2026-08-11T16:43:44.149261Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.06082","last_updated":"2024-02-08T22:17:40Z","snapshot_observed_at":"2026-08-15T04:48:44.897402Z","submitted_at":"2024-02-08T22:17:40Z","title":"SubGen: Token Generation in Sublinear Time and Memory","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.06082","snapshot_observed_at":"2026-08-11T16:43:44.153690Z","title":"Subgen: Token generation in sublinear time and memory","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:44.153690Z"},"links":{"cited_paper":"/paper/2402.06082","citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:fefd1d50b904632f65e343e26d473fb901c6222c8dfe8770f84435ed7db5d995","observation_id":"5ad57403-d026-4087-add5-b4b343916b03","resolution":{"observed_at":"2026-08-11T16:43:44.153690Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1911.02150","last_updated":"2019-11-06T00:19:05Z","snapshot_observed_at":"2026-07-06T08:35:01.386074Z","submitted_at":"2019-11-06T00:19:05Z","title":"Fast Transformer Decoding: One Write-Head is All You Need","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1911.02150","snapshot_observed_at":"2026-08-11T16:43:44.157913Z","title":"Fast transformer decoding: One write-head is all you need","venue":null,"work_id":null,"year":1911},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:44.157913Z"},"links":{"cited_paper":"/paper/1911.02150","citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:f4a8770ba33007eb4b188548366e4876c8237c71ee29cbc9723e964683d1dff5","observation_id":"50d55882-1fdf-4ad4-af3e-05cd0c81f054","resolution":{"observed_at":"2026-08-11T16:43:44.157913Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.18079","last_updated":"2025-05-28T18:58:29Z","snapshot_observed_at":"2026-08-13T04:29:50.330115Z","submitted_at":"2024-01-31T18:58:14Z","title":"KVQuant: Towards 10 Million Context Length LLM Inference with KV Cache Quantization","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.18079","snapshot_observed_at":"2026-08-11T16:43:44.162332Z","title":"Kvquant: Towards 10 million context length llm inference with kv cache quantization","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:44.162332Z"},"links":{"cited_paper":"/paper/2401.18079","citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:bf2e93010b0d141424a114f2fbc2ac93a697997cd338df6eae91a29230a0bc13","observation_id":"66bd7130-778b-4b9b-b5f9-bb1cb30f5782","resolution":{"observed_at":"2026-08-11T16:43:44.162332Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:43:44.884667Z","title":"Flexgen: High-throughput generative inference of large language models with a single gpu","venue":null,"work_id":"fb588999-eae7-4848-a465-8c673801ed45","year":2023},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:44.166732Z"},"links":{"citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:aa43350c63837fd26a45725d4ad7e7e57bd74addd17bd7e036729a5ea950204e","observation_id":"44d98f3e-24f5-472d-9ddd-3fe1da577ed0","resolution":{"observed_at":"2026-08-11T16:43:44.891314Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:43:44.170735Z","title":"Transformers are rnns: Fast autoregressive transformers with linear attention","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:44.170735Z"},"links":{"citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:9ec0b1cd2cc5ba21ca8847102c517c92c97ebc4f98c039cfe2362e22df6452fc","observation_id":"e5f23e26-9437-44f2-b0a4-a624e0d6bf93","resolution":{"observed_at":"2026-08-11T16:43:44.170735Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.01876","last_updated":"2024-03-04T09:32:05Z","snapshot_observed_at":"2026-08-14T02:05:56.579091Z","submitted_at":"2024-03-04T09:32:05Z","title":"D\\'ej\\`aVu: KV-cache Streaming for Fast, Fault-tolerant Generative LLM Serving","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.01876","snapshot_observed_at":"2026-08-11T16:43:44.201438Z","title":"D 'ej avu: Kv-cache streaming for fast, fault-tolerant generative llm serving","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:44.201438Z"},"links":{"cited_paper":"/paper/2403.01876","citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:f11c270a0e26d4829aa9dc428a5cc300074099df22c09ff0a1908b2e88603246","observation_id":"5fe69c5c-7bc4-4766-9cf6-10f572dab38b","resolution":{"observed_at":"2026-08-11T16:43:44.201438Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:43:44.246679Z","title":"What disease does this patient have? a large-scale open domain question answering dataset from medical exams","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:44.246679Z"},"links":{"citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:e2fd90003911b7fab1d352755ad30b4e1d89b462ddfc501d9d4b4b54798cf71c","observation_id":"4c556cb1-c18e-4442-97b0-f8adf25b3dc3","resolution":{"observed_at":"2026-08-11T16:43:44.246679Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17453","last_updated":"2024-04-07T00:56:53Z","snapshot_observed_at":"2026-08-14T06:01:54.549199Z","submitted_at":"2023-09-29T17:59:56Z","title":"Efficient Streaming Language Models with Attention Sinks","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.17453","snapshot_observed_at":"2026-08-11T16:43:44.266533Z","title":"Efficient streaming language models with attention sinks","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:44.266533Z"},"links":{"cited_paper":"/paper/2309.17453","citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:a555522806a25f5e490f1c3e921cba2b5b46aa68312c5b95a190ebc9df8dae7d","observation_id":"18bfa255-40bd-434b-9c8e-78b584cb5fb7","resolution":{"observed_at":"2026-08-11T16:43:44.266533Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1904.10509","last_updated":"2019-04-23T19:29:47Z","snapshot_observed_at":"2026-08-09T19:46:04.857927Z","submitted_at":"2019-04-23T19:29:47Z","title":"Generating Long Sequences with Sparse Transformers","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.10509","snapshot_observed_at":"2026-08-11T16:43:44.304635Z","title":"Generating long sequences with sparse transformers","venue":null,"work_id":null,"year":1904},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:44.304635Z"},"links":{"cited_paper":"/paper/1904.10509","citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:fb2d49ede7f8c52249f836740d3054751405640190fcf22be1d7d7d19df51952","observation_id":"aab0e56d-570c-47d0-9efb-655009d2b5e4","resolution":{"observed_at":"2026-08-11T16:43:44.304635Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2004.05150","last_updated":"2020-12-02T17:52:35Z","snapshot_observed_at":"2026-07-31T17:17:17.205582Z","submitted_at":"2020-04-10T17:54:09Z","title":"Longformer: The Long-Document Transformer","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2004.05150","snapshot_observed_at":"2026-08-11T16:43:44.392915Z","title":"Longformer: The long-document transformer","venue":null,"work_id":null,"year":2004},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:44.392915Z"},"links":{"cited_paper":"/paper/2004.05150","citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:09019ec445354f9c0d4145a27191ecd9d1cc3ed574b70d84c400d1351ed6f916","observation_id":"481a80fe-1da7-4bcb-9149-bf84ce27f53a","resolution":{"observed_at":"2026-08-11T16:43:44.392915Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.06899","last_updated":"2021-06-13T02:30:23Z","snapshot_observed_at":"2026-08-13T19:04:40.553642Z","submitted_at":"2021-06-13T02:30:23Z","title":"Memory-efficient Transformers via Top-$k$ Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.06899","snapshot_observed_at":"2026-08-11T16:43:44.403049Z","title":"Memory-efficient transformers via top- k attention","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","version":3},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-11T16:43:44.403049Z"},"links":{"cited_paper":"/paper/2106.06899","citing_paper":"/paper/2412.16187"},"observation_digest":"sha256:fc28f327cc4f3655ae5a7d1e9b065b3aad60ddf84b0c50c0ac962dc77eaf86e9","observation_id":"a80dc761-5a64-46f3-98dd-5e7ac4511eb9","resolution":{"observed_at":"2026-08-11T16:43:44.403049Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2412.16187","last_updated":"2025-06-04T22:37:29Z","latest_version":3,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-14T11:46:16.870794Z","submitted_at":"2024-12-13T06:00:27Z","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing"},"reference_resolution":{"displayed":39,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":32,"verified_exact":0,"verified_fuzzy":7},"total_outbound_references":39},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 16 August 2026, this Paper Citation Record lists 39 of 39 outbound references and 1 inbound Pith citation observation for arXiv:2412.16187."}