{"as_of":"2026-08-10T02:17:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:5a13a117fa8a9d1e01af53e3de83a7a9f3122576e2db6f51c869bc4cca041d98","coverage":[{"denominator":9,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":9,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T02:10:50.027276Z","state":"measured"},{"denominator":9,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":9,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.14431/citation-record","integrity":"/paper/2607.14431/integrity","json":"/paper/2607.14431/citation-record.json","paper":"/paper/2607.14431"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2310.07240","last_updated":"2024-07-19T21:04:14Z","snapshot_observed_at":"2026-08-07T17:46:33.336201Z","submitted_at":"2023-10-11T07:08:20Z","title":"CacheGen: KV Cache Compression and Streaming for Fast Large Language Model Serving","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.07240","snapshot_observed_at":"2026-08-02T02:10:49.534551Z","title":"LMCache: An Eﬀicient KV Cache Layer for Enterprise-Scale LLM Inference","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.14431","last_updated":"2026-07-15T23:55:48Z","snapshot_observed_at":"2026-08-07T13:00:58.960116Z","submitted_at":"2026-07-15T23:55:48Z","title":"Smarter and Cheaper at Once: Byte-Exact KV-Cache Grafting Turns a Frozen Small Model into a Verified-Knowledge Flywheel","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-02T02:10:49.534551Z"},"links":{"cited_paper":"/paper/2310.07240","citing_paper":"/paper/2607.14431"},"observation_digest":"sha256:47b782492a18e6b8107375dad0cc90ce6930625323f4b320903901dba4e42c88","observation_id":"dcc0f714-5a6a-48f1-8a67-3630cf0085b9","resolution":{"observed_at":"2026-08-02T02:10:49.534551Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.11171","last_updated":"2023-03-07T17:57:37Z","snapshot_observed_at":"2026-07-06T12:50:22.773056Z","submitted_at":"2022-03-21T17:48:52Z","title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.11171","snapshot_observed_at":"2026-08-02T02:10:49.822456Z","title":"LiveBench: A Challenging, Contamination-Free LLM Benchmark","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.14431","last_updated":"2026-07-15T23:55:48Z","snapshot_observed_at":"2026-08-07T13:00:58.960116Z","submitted_at":"2026-07-15T23:55:48Z","title":"Smarter and Cheaper at Once: Byte-Exact KV-Cache Grafting Turns a Frozen Small Model into a Verified-Knowledge Flywheel","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-02T02:10:49.822456Z"},"links":{"cited_paper":"/paper/2203.11171","citing_paper":"/paper/2607.14431"},"observation_digest":"sha256:67472593834af1b66e592f4f59653158731fb630230db8f1f951984d8e387e0d","observation_id":"c9ca5997-fc7b-41b6-8a17-a805ad298c9b","resolution":{"observed_at":"2026-08-02T02:10:49.822456Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.16444","last_updated":"2025-04-03T22:49:22Z","snapshot_observed_at":"2026-08-07T00:11:38.883613Z","submitted_at":"2024-05-26T06:00:17Z","title":"CacheBlend: Fast Large Language Model Serving for RAG with Cached Knowledge Fusion","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.16444","snapshot_observed_at":"2026-08-02T02:10:49.918894Z","title":"CacheBlend: Fast Large Language Model Serving for RAG with Cached Knowledge Fusion","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14431","last_updated":"2026-07-15T23:55:48Z","snapshot_observed_at":"2026-08-07T13:00:58.960116Z","submitted_at":"2026-07-15T23:55:48Z","title":"Smarter and Cheaper at Once: Byte-Exact KV-Cache Grafting Turns a Frozen Small Model into a Verified-Knowledge Flywheel","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-02T02:10:49.918894Z"},"links":{"cited_paper":"/paper/2405.16444","citing_paper":"/paper/2607.14431"},"observation_digest":"sha256:f8a370457f890d2c75c5411d75d9e11cc747abf541c48bb5438b60adfa5c911d","observation_id":"e25a3900-d4c9-4931-848e-d733724b97b2","resolution":{"observed_at":"2026-08-02T02:10:49.918894Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.07104","last_updated":"2024-06-06T00:10:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-12T09:34:27Z","title":"SGLang: Efficient Execution of Structured Language Model Programs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.07104","snapshot_observed_at":"2026-08-02T02:10:50.027276Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.14431","last_updated":"2026-07-15T23:55:48Z","snapshot_observed_at":"2026-08-07T13:00:58.960116Z","submitted_at":"2026-07-15T23:55:48Z","title":"Smarter and Cheaper at Once: Byte-Exact KV-Cache Grafting Turns a Frozen Small Model into a Verified-Knowledge Flywheel","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-02T02:10:50.027276Z"},"links":{"cited_paper":"/paper/2312.07104","citing_paper":"/paper/2607.14431"},"observation_digest":"sha256:bb3d4e41bfa3ce37dbaa25d4066ba53c174359da64840831daeff1bd4fe443fd","observation_id":"07fac811-4fb5-4a0f-974c-5002f3296a5c","resolution":{"observed_at":"2026-08-02T02:10:50.027276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T02:10:49.709223Z","title":"Energy and Policy Considerations for Deep Learning in NLP","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2607.14431","last_updated":"2026-07-15T23:55:48Z","snapshot_observed_at":"2026-08-07T13:00:58.960116Z","submitted_at":"2026-07-15T23:55:48Z","title":"Smarter and Cheaper at Once: Byte-Exact KV-Cache Grafting Turns a Frozen Small Model into a Verified-Knowledge Flywheel","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-02T02:10:49.709223Z"},"links":{"citing_paper":"/paper/2607.14431"},"observation_digest":"sha256:601bd407225b74150e92bdde716df5bbc6d1e194b6359e9732e94c3ce3400449","observation_id":"2fd66561-b457-4214-926e-88c6a9ab89a0","resolution":{"observed_at":"2026-08-02T02:10:49.709223Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06180","last_updated":"2023-09-12T12:50:04Z","snapshot_observed_at":"2026-08-02T09:51:08.145755Z","submitted_at":"2023-09-12T12:50:04Z","title":"Efficient Memory Management for Large Language Model Serving with PagedAttention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.06180","snapshot_observed_at":"2026-08-02T02:10:49.471962Z","title":"Quantifying the Carbon Emissions of Machine Learning","venue":null,"work_id":null,"year":1910},"citing_paper":{"arxiv_id":"2607.14431","last_updated":"2026-07-15T23:55:48Z","snapshot_observed_at":"2026-08-07T13:00:58.960116Z","submitted_at":"2026-07-15T23:55:48Z","title":"Smarter and Cheaper at Once: Byte-Exact KV-Cache Grafting Turns a Frozen Small Model into a Verified-Knowledge Flywheel","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-02T02:10:49.471962Z"},"links":{"cited_paper":"/paper/2309.06180","citing_paper":"/paper/2607.14431"},"observation_digest":"sha256:88dede349c9e6164e3325efe95b852c52acb2909b3ba4aba072c6f534859d3be","observation_id":"156c3498-366a-4eb3-8ae2-12a5cb9308ce","resolution":{"observed_at":"2026-08-02T02:10:49.471962Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04934","last_updated":"2024-04-25T15:45:19Z","snapshot_observed_at":"2026-07-06T16:44:55.494764Z","submitted_at":"2023-11-07T18:17:05Z","title":"Prompt Cache: Modular Attention Reuse for Low-Latency Inference","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.04934","snapshot_observed_at":"2026-08-02T02:10:49.354541Z","title":"Gemma 4 model card","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.14431","last_updated":"2026-07-15T23:55:48Z","snapshot_observed_at":"2026-08-07T13:00:58.960116Z","submitted_at":"2026-07-15T23:55:48Z","title":"Smarter and Cheaper at Once: Byte-Exact KV-Cache Grafting Turns a Frozen Small Model into a Verified-Knowledge Flywheel","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-02T02:10:49.354541Z"},"links":{"cited_paper":"/paper/2311.04934","citing_paper":"/paper/2607.14431"},"observation_digest":"sha256:63553cc7d1be9cc8fcc6ad602355c5b86fa0e90002e4ae0cbcfb186d9db4da88","observation_id":"a13ebabe-ba70-4929-aba6-63c486ffc31a","resolution":{"observed_at":"2026-08-02T02:10:49.354541Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.09990","last_updated":"2026-05-11T05:06:59Z","snapshot_observed_at":"2026-07-06T23:21:59.096464Z","submitted_at":"2026-05-11T05:06:59Z","title":"Merlin: Deterministic Byte-Exact Deduplication for Lossless Context Optimization in Large Language Model Inference","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.09990","snapshot_observed_at":"2026-08-02T02:10:49.607668Z","title":"Qwen3.6-35B-A3B","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.14431","last_updated":"2026-07-15T23:55:48Z","snapshot_observed_at":"2026-08-07T13:00:58.960116Z","submitted_at":"2026-07-15T23:55:48Z","title":"Smarter and Cheaper at Once: Byte-Exact KV-Cache Grafting Turns a Frozen Small Model into a Verified-Knowledge Flywheel","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-02T02:10:49.607668Z"},"links":{"cited_paper":"/paper/2605.09990","citing_paper":"/paper/2607.14431"},"observation_digest":"sha256:8903824ea22da1c120d0a141997a97b08670633e97583a24a6cd4d14cc4e2c16","observation_id":"0d19c8ef-1365-4605-88ae-b3ed6e986b9f","resolution":{"observed_at":"2026-08-02T02:10:49.607668Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2606.13097","last_updated":"2026-07-29T12:55:33Z","snapshot_observed_at":"2026-08-06T14:19:46.913836Z","submitted_at":"2026-06-11T09:25:27Z","title":"Functional Cache Grafting: Robust and Rapid Code-Policy Synthesis for Embodied Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2606.13097","snapshot_observed_at":"2026-08-02T02:10:49.265127Z","title":"Spanner: Google’s Globally-Distributed Database","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2607.14431","last_updated":"2026-07-15T23:55:48Z","snapshot_observed_at":"2026-08-07T13:00:58.960116Z","submitted_at":"2026-07-15T23:55:48Z","title":"Smarter and Cheaper at Once: Byte-Exact KV-Cache Grafting Turns a Frozen Small Model into a Verified-Knowledge Flywheel","version":1},"reference_index":2026,"source":"pdf_text","source_observed_at":"2026-08-02T02:10:49.265127Z"},"links":{"cited_paper":"/paper/2606.13097","citing_paper":"/paper/2607.14431"},"observation_digest":"sha256:4700ecc2d45366e70d34d0d74e716ecfc357588cc37644eec0d59082e0c9b498","observation_id":"cfbb21ea-747d-40fe-ae00-09e9c6d68dde","resolution":{"observed_at":"2026-08-02T02:10:49.265127Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.14431","last_updated":"2026-07-15T23:55:48Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-07T13:00:58.960116Z","submitted_at":"2026-07-15T23:55:48Z","title":"Smarter and Cheaper at Once: Byte-Exact KV-Cache Grafting Turns a Frozen Small Model into a Verified-Knowledge Flywheel"},"reference_resolution":{"displayed":9,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":9,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":9},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 9 of 9 outbound references and 0 inbound Pith citation observations for arXiv:2607.14431."}