{"as_of":"2026-08-09T06:55:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4df515dbf0660fb610dcba572bf21135870f5ee85756743e7c047cb7a01bb9eb","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":8,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":8,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":8,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":8,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T10:10:23.098928Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-06-30T17:24:56.959321Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2410.00428","last_updated":"2024-10-09T11:40:31Z","snapshot_observed_at":"2026-07-06T19:25:06.162911Z","submitted_at":"2024-10-01T06:23:17Z","title":"LayerKV: Optimizing Large Language Model Serving with Layer-wise KV Cache Management","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.00428","snapshot_observed_at":"2026-08-08T10:10:23.098928Z","title":"Xiong, H","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08182","last_updated":"2025-02-12T07:42:45Z","snapshot_observed_at":"2026-08-08T14:36:12.899386Z","submitted_at":"2025-02-12T07:42:45Z","title":"Memory Offloading for Large Language Model Inference with Latency SLO Guarantees","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-08T10:10:23.098928Z"},"links":{"cited_paper":"/paper/2410.00428","citing_paper":"/paper/2502.08182"},"observation_digest":"sha256:7c36e3f4edf46303ef5aaa9a5615c53dc7e5734607451df028fb7d064e1687e4","observation_id":"702c9932-288c-49ce-8502-41fabbff8863","resolution":{"observed_at":"2026-08-08T10:10:23.098928Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.00428","last_updated":"2024-10-09T11:40:31Z","snapshot_observed_at":"2026-07-06T19:25:06.162911Z","submitted_at":"2024-10-01T06:23:17Z","title":"LayerKV: Optimizing Large Language Model Serving with Layer-wise KV Cache Management","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.00428","snapshot_observed_at":"2026-08-07T06:03:19.359687Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.11104","last_updated":"2025-06-06T20:24:36Z","snapshot_observed_at":"2026-08-07T20:41:32.031099Z","submitted_at":"2025-06-06T20:24:36Z","title":"DAM: Dynamic Attention Mask for Long-Context Large Language Model Inference Acceleration","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-07T06:03:19.359687Z"},"links":{"cited_paper":"/paper/2410.00428","citing_paper":"/paper/2506.11104"},"observation_digest":"sha256:6df4a224d6d748314fd9c5143039d7653e84e918f27840f2daba4d3cbc72c7c7","observation_id":"f770c5da-57e3-4456-a076-a60800a339bf","resolution":{"observed_at":"2026-08-07T06:03:19.359687Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.00428","last_updated":"2024-10-09T11:40:31Z","snapshot_observed_at":"2026-07-06T19:25:06.162911Z","submitted_at":"2024-10-01T06:23:17Z","title":"LayerKV: Optimizing Large Language Model Serving with Layer-wise KV Cache Management","version":3},"cited_work":{"arxiv_id":"2410.00428","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.00428","snapshot_observed_at":"2026-06-30T17:24:56.959321Z","title":"Layerkv: Optimizing large language model serving with layer-wise kv cache management","venue":null,"work_id":"7e52dd1f-0450-4444-b4c0-b7724d2618af","year":2024},"citing_paper":{"arxiv_id":"2602.09725","last_updated":"2026-05-12T14:12:11Z","snapshot_observed_at":"2026-07-06T22:45:15.978102Z","submitted_at":"2026-02-10T12:29:02Z","title":"Efficient Remote KV Cache Reuse with GPU-native Video Codec","version":3},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-05-16T05:21:04.555356Z"},"links":{"cited_paper":"/paper/2410.00428","citing_paper":"/paper/2602.09725"},"observation_digest":"sha256:b580a3c8731fe8ab535155fbc8c7fbbd409131a1ceea1fdaf37016caaba2995b","observation_id":"bca2aa03-58fd-45a2-adcc-31e71317e029","resolution":{"observed_at":"2026-05-16T05:22:22.824323Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.00428","last_updated":"2024-10-09T11:40:31Z","snapshot_observed_at":"2026-07-06T19:25:06.162911Z","submitted_at":"2024-10-01T06:23:17Z","title":"LayerKV: Optimizing Large Language Model Serving with Layer-wise KV Cache Management","version":3},"cited_work":{"arxiv_id":"2410.00428","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.00428","snapshot_observed_at":"2026-06-30T17:24:56.959321Z","title":"Layerkv: Optimizing large language model serving with layer-wise kv cache management","venue":null,"work_id":"7e52dd1f-0450-4444-b4c0-b7724d2618af","year":2024},"citing_paper":{"arxiv_id":"2604.06370","last_updated":"2026-04-07T18:52:25Z","snapshot_observed_at":"2026-08-08T21:24:58.736715Z","submitted_at":"2026-04-07T18:52:25Z","title":"ForkKV: Scaling Multi-LoRA Agent Serving via Copy-on-Write Disaggregated KV Cache","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-10T18:16:49.292491Z"},"links":{"cited_paper":"/paper/2410.00428","citing_paper":"/paper/2604.06370"},"observation_digest":"sha256:b91e6010c71e6391f0190bb4b2d0fc211aa5a08c675cd690ede2b845cab842fc","observation_id":"0b7c41ff-21a0-42bd-b047-83131df9db7a","resolution":{"observed_at":"2026-05-11T05:10:54.903875Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.00428","last_updated":"2024-10-09T11:40:31Z","snapshot_observed_at":"2026-07-06T19:25:06.162911Z","submitted_at":"2024-10-01T06:23:17Z","title":"LayerKV: Optimizing Large Language Model Serving with Layer-wise KV Cache Management","version":3},"cited_work":{"arxiv_id":"2410.00428","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.00428","snapshot_observed_at":"2026-06-30T17:24:56.959321Z","title":"Layerkv: Optimizing large language model serving with layer-wise kv cache management","venue":null,"work_id":"7e52dd1f-0450-4444-b4c0-b7724d2618af","year":2024},"citing_paper":{"arxiv_id":"2605.24022","last_updated":"2026-05-20T08:59:48Z","snapshot_observed_at":"2026-08-02T23:31:27.025065Z","submitted_at":"2026-05-20T08:59:48Z","title":"Adaptive KV Cache Reuse for Fast Long-Context LLM Serving","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-30T17:23:13.154458Z"},"links":{"cited_paper":"/paper/2410.00428","citing_paper":"/paper/2605.24022"},"observation_digest":"sha256:7bed4503b63216c872186eb5f8b0baf3ab93540bc65d8e1feb8ff0734c6e7100","observation_id":"92e32207-aca9-4950-81da-403be932bc1d","resolution":{"observed_at":"2026-06-30T17:24:56.960758Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.00428","last_updated":"2024-10-09T11:40:31Z","snapshot_observed_at":"2026-07-06T19:25:06.162911Z","submitted_at":"2024-10-01T06:23:17Z","title":"LayerKV: Optimizing Large Language Model Serving with Layer-wise KV Cache Management","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.00428","snapshot_observed_at":"2026-08-01T21:09:10.825037Z","title":"LayerKV: Optimizing large language model serving with layer-wise KV cache management, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.16184","last_updated":"2026-07-17T17:58:29Z","snapshot_observed_at":"2026-08-05T03:58:48.412708Z","submitted_at":"2026-07-17T17:58:29Z","title":"PagedWeight: Efficient MoE LLM Serving with Dynamic Quality-Aware Weight Quantization","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:10.825037Z"},"links":{"cited_paper":"/paper/2410.00428","citing_paper":"/paper/2607.16184"},"observation_digest":"sha256:d0de93727bf0b2fdb3729649ade51c1087268a5474bfe61b7c8b8f8b18c5d521","observation_id":"baea7759-e46b-49ec-9f53-6230a91772e4","resolution":{"observed_at":"2026-08-01T21:09:10.825037Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.00428","last_updated":"2024-10-09T11:40:31Z","snapshot_observed_at":"2026-07-06T19:25:06.162911Z","submitted_at":"2024-10-01T06:23:17Z","title":"LayerKV: Optimizing Large Language Model Serving with Layer-wise KV Cache Management","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.00428","snapshot_observed_at":"2026-08-01T19:51:22.631756Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.16836","last_updated":"2026-07-27T11:12:03Z","snapshot_observed_at":"2026-08-09T06:54:03.941605Z","submitted_at":"2026-07-18T14:21:39Z","title":"Beyond Storage: State as a Runtime Control Problem in Parallel and Distributed Systems","version":2},"reference_index":230,"source":"pdf_text","source_observed_at":"2026-08-01T19:51:22.631756Z"},"links":{"cited_paper":"/paper/2410.00428","citing_paper":"/paper/2607.16836"},"observation_digest":"sha256:c26d96222206825b90ae1481941127bb1456dc7f8524d5e6ddf75b2573b58875","observation_id":"2a1b2494-8811-4498-b14e-4b0f6b4fe1db","resolution":{"observed_at":"2026-08-01T19:51:22.631756Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.00428","last_updated":"2024-10-09T11:40:31Z","snapshot_observed_at":"2026-07-06T19:25:06.162911Z","submitted_at":"2024-10-01T06:23:17Z","title":"LayerKV: Optimizing Large Language Model Serving with Layer-wise KV Cache Management","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.00428","snapshot_observed_at":"2026-08-01T07:46:27.035461Z","title":"LayerKV: Optimizing Large Language Model Serving with Layer-wise KV Cache Management","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.21686","last_updated":"2026-07-23T14:39:46Z","snapshot_observed_at":"2026-08-08T08:20:20.803685Z","submitted_at":"2026-07-23T14:39:46Z","title":"Persistent Computational State: A Session-Centric Runtime for Generative World Models","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-01T07:46:27.035461Z"},"links":{"cited_paper":"/paper/2410.00428","citing_paper":"/paper/2607.21686"},"observation_digest":"sha256:2be6b10f180f0f02a4cb57bd7f3058fc1875ac152a235d69e3ff1135e03281f5","observation_id":"d91cb980-1b46-4f98-b2e4-c768f8990f3f","resolution":{"observed_at":"2026-08-01T07:46:27.035461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2410.00428/citation-record","integrity":"/paper/2410.00428/integrity","json":"/paper/2410.00428/citation-record.json","paper":"/paper/2410.00428"},"outbound":[],"paper":{"arxiv_id":"2410.00428","last_updated":"2024-10-09T11:40:31Z","latest_version":3,"primary_category":"cs.DC","snapshot_observed_at":"2026-07-06T19:25:06.162911Z","submitted_at":"2024-10-01T06:23:17Z","title":"LayerKV: Optimizing Large Language Model Serving with Layer-wise KV Cache Management"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 8 inbound Pith citation observations for arXiv:2410.00428."}