{"as_of":"2026-08-09T14:10:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:92a0b05700cab6d4a451fb8d8265473ceed5a672c555950071b0e4583658f01d","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":29,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":29,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":29,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":29,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T04:32:01.202215Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T09:39:46.803577Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":"2406.17565","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-07-04T09:39:46.803577Z","title":"Memserve: Con- text caching for disaggregated llm serving with elastic memory pool.arXiv preprint arXiv:2406.17565","venue":null,"work_id":"51c7ec72-def3-42da-9410-c0152debc67f","year":2024},"citing_paper":{"arxiv_id":"2412.03594","last_updated":"2026-04-22T15:33:51Z","snapshot_observed_at":"2026-07-06T20:01:45.826971Z","submitted_at":"2024-11-29T05:57:37Z","title":"BatchLLM: Optimizing Large Batched LLM Inference with Global Prefix Sharing and Throughput-oriented Token Batching","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-23T16:57:46.645061Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2412.03594"},"observation_digest":"sha256:fc45f4a314342d55f3574f350dde8db96f608855cc422c34b1ecc5e64862d330","observation_id":"3b653b39-854e-4c09-bc2c-fb4c64f5f2fc","resolution":{"observed_at":"2026-05-23T16:58:11.904020Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-08-09T04:32:01.202215Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.03589","last_updated":"2025-02-05T20:09:51Z","snapshot_observed_at":"2026-08-09T04:23:35.804076Z","submitted_at":"2025-02-05T20:09:51Z","title":"HACK: Homomorphic Acceleration via Compression of the Key-Value Cache for Disaggregated LLM Inference","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-09T04:32:01.202215Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2502.03589"},"observation_digest":"sha256:20c090a18ed2637cb6486e9cdf4102b02e20a6a41b14756ae9b75a748d8f3a88","observation_id":"08a01091-abe2-435f-a039-308069137136","resolution":{"observed_at":"2026-08-09T04:32:01.202215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":"2406.17565","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-07-04T09:39:46.803577Z","title":"Memserve: Con- text caching for disaggregated llm serving with elastic memory pool.arXiv preprint arXiv:2406.17565","venue":null,"work_id":"51c7ec72-def3-42da-9410-c0152debc67f","year":2024},"citing_paper":{"arxiv_id":"2504.15965","last_updated":"2025-04-23T13:47:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-22T15:05:04Z","title":"From Human Memory to AI Memory: A Survey on Memory Mechanisms in the Era of LLMs","version":2},"reference_index":122,"source":"pdf_text","source_observed_at":"2026-05-17T11:05:09.588491Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2504.15965"},"observation_digest":"sha256:a4daa8679e5c464735415f2a28a78a7184820fcb69ae403b1fe32ac5db1f7f4d","observation_id":"0b30afa7-a920-4a3a-92ee-b8aab4eef3cd","resolution":{"observed_at":"2026-05-17T11:05:09.848606Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-08-07T14:47:55.773082Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17694","last_updated":"2026-03-28T10:14:51Z","snapshot_observed_at":"2026-08-07T14:40:02.904487Z","submitted_at":"2025-05-23T10:03:28Z","title":"CoDec: Prefix-Shared Decoding Kernel for LLMs","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:55.773082Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2505.17694"},"observation_digest":"sha256:3d2ce5717957c243c95491075f87de84dffa823a748debfe4933bf6c205307cf","observation_id":"943e6b8a-7059-48d4-b290-02b90832b38b","resolution":{"observed_at":"2026-08-07T14:47:55.773082Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-08-07T10:26:32.777159Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05508","last_updated":"2025-06-05T18:47:49Z","snapshot_observed_at":"2026-08-07T13:07:21.433278Z","submitted_at":"2025-06-05T18:47:49Z","title":"Beyond the Buzz: A Pragmatic Take on Inference Disaggregation","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T10:26:32.777159Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2506.05508"},"observation_digest":"sha256:d89ccd9789d37a5dd391d45bb2fc1ba189a0e41e9e925071523bb33bc83db4f0","observation_id":"3f1b93ad-e1b0-450a-ac38-19ff0a11b44e","resolution":{"observed_at":"2026-08-07T10:26:32.777159Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-08-07T06:03:19.319925Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.11104","last_updated":"2025-06-06T20:24:36Z","snapshot_observed_at":"2026-08-07T20:41:32.031099Z","submitted_at":"2025-06-06T20:24:36Z","title":"DAM: Dynamic Attention Mask for Long-Context Large Language Model Inference Acceleration","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-07T06:03:19.319925Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2506.11104"},"observation_digest":"sha256:0750246203eefe5a9205b939da423cf9ec365aa19553109f994bed4bc2cbdf11","observation_id":"964a02ee-e3b0-467c-b77d-b3cd16ae2297","resolution":{"observed_at":"2026-08-07T06:03:19.319925Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-08-05T16:21:09.133223Z","title":"arXiv:2406.17565","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.18736","last_updated":"2025-08-26T07:09:09Z","snapshot_observed_at":"2026-08-09T04:17:30.728713Z","submitted_at":"2025-08-26T07:09:09Z","title":"Rethinking Caching for LLM Serving Systems: Beyond Traditional Heuristics","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-05T16:21:09.133223Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2508.18736"},"observation_digest":"sha256:fee9f12f5f9e8d090109e2549a9503f66595e31e1cd64bf070df755cebd7e4c8","observation_id":"cb7ffe8d-b83e-4d3b-9d8f-7859cb9a33bf","resolution":{"observed_at":"2026-08-05T16:21:09.133223Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-08-05T15:45:36.304749Z","title":"Mem- serve: Flexible mem pool for building disaggre- gated llm serving with caching, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.19559","last_updated":"2025-08-27T04:22:02Z","snapshot_observed_at":"2026-08-08T02:10:39.123561Z","submitted_at":"2025-08-27T04:22:02Z","title":"Taming the Chaos: Coordinated Autoscaling for Heterogeneous and Disaggregated LLM Inference","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-05T15:45:36.304749Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2508.19559"},"observation_digest":"sha256:57790b7a1e1020f11b7b49a015c997f1159aeb22399be4735311605bf4de0175","observation_id":"3b62d94d-a129-4d69-9b26-5a56356e42ca","resolution":{"observed_at":"2026-08-05T15:45:36.304749Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-08-05T15:26:33.726798Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.19870","last_updated":"2025-08-27T13:33:35Z","snapshot_observed_at":"2026-08-07T05:20:03.066438Z","submitted_at":"2025-08-27T13:33:35Z","title":"Secure Multi-LLM Agentic AI and Agentification for Edge General Intelligence by Zero-Trust: A Survey","version":1},"reference_index":121,"source":"pdf_text","source_observed_at":"2026-08-05T15:26:33.726798Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2508.19870"},"observation_digest":"sha256:e9a37d3f057804a48daa6f90e34c7f21a2fbc5787b3bd5ef39bfb3c7cbc9766e","observation_id":"26143fc2-cb77-4548-87f0-3d769c800d34","resolution":{"observed_at":"2026-08-05T15:26:33.726798Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-08-04T18:58:12.680361Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09560","last_updated":"2025-09-11T15:51:43Z","snapshot_observed_at":"2026-08-06T18:46:16.977725Z","submitted_at":"2025-09-11T15:51:43Z","title":"Boosting Embodied AI Agents through Perception-Generation Disaggregation and Asynchronous Pipeline Execution","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-04T18:58:12.680361Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2509.09560"},"observation_digest":"sha256:76b0fd2b1e4f60be6a4a4dec418fc261fa822e94f8af4a0398cf5af8fcc451a2","observation_id":"26a19a2d-3907-4e66-92b0-bba1b5cab032","resolution":{"observed_at":"2026-08-04T18:58:12.680361Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-08-03T23:39:06.591452Z","title":"Memserve: Context caching for disaggregated llm serving with elastic memory pool, 2024a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2511.04791","last_updated":"2026-05-30T23:27:43Z","snapshot_observed_at":"2026-08-03T23:39:04.093171Z","submitted_at":"2025-11-06T20:18:34Z","title":"DuetServe: Harmonizing Prefill and Decode for LLM Serving via Adaptive GPU Multiplexing","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-03T23:39:06.591452Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2511.04791"},"observation_digest":"sha256:a4638f7459498b2618163c9496e90acde420bc47282de14ee56c0faeea682be4","observation_id":"ae877024-34f8-43a3-9549-b818ea3d9c92","resolution":{"observed_at":"2026-08-03T23:39:06.591452Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":"2406.17565","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-07-04T09:39:46.803577Z","title":"Memserve: Con- text caching for disaggregated llm serving with elastic memory pool.arXiv preprint arXiv:2406.17565","venue":null,"work_id":"51c7ec72-def3-42da-9410-c0152debc67f","year":2024},"citing_paper":{"arxiv_id":"2601.20309","last_updated":"2026-05-18T19:51:16Z","snapshot_observed_at":"2026-07-06T22:43:17.026472Z","submitted_at":"2026-01-28T07:01:46Z","title":"SuperInfer: SLO-Aware Rotary Scheduling and Memory Management for LLM Inference on Superchips","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-21T15:26:01.283448Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2601.20309"},"observation_digest":"sha256:6e72dc13576bb2f023bb9b31bca378342149292d3512e6a3880a83978a71f4b8","observation_id":"4066d4c7-8537-4b24-91ea-a2cc0bb9a66a","resolution":{"observed_at":"2026-05-21T15:30:17.918779Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":"2406.17565","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-07-04T09:39:46.803577Z","title":"Memserve: Con- text caching for disaggregated llm serving with elastic memory pool.arXiv preprint arXiv:2406.17565","venue":null,"work_id":"51c7ec72-def3-42da-9410-c0152debc67f","year":2024},"citing_paper":{"arxiv_id":"2602.09725","last_updated":"2026-05-12T14:12:11Z","snapshot_observed_at":"2026-07-06T22:45:15.978102Z","submitted_at":"2026-02-10T12:29:02Z","title":"Efficient Remote KV Cache Reuse with GPU-native Video Codec","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-16T05:21:04.555356Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2602.09725"},"observation_digest":"sha256:1cea382c1e483a919a282f8cd493119fe5fe7372dcf383b407b59ba9db52a5f3","observation_id":"47bd091b-30b8-416a-a99f-bbff14be2f8f","resolution":{"observed_at":"2026-05-16T05:22:22.835574Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":"2406.17565","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-07-04T09:39:46.803577Z","title":"Memserve: Con- text caching for disaggregated llm serving with elastic memory pool.arXiv preprint arXiv:2406.17565","venue":null,"work_id":"51c7ec72-def3-42da-9410-c0152debc67f","year":2024},"citing_paper":{"arxiv_id":"2603.27960","last_updated":"2026-04-11T06:07:02Z","snapshot_observed_at":"2026-07-06T22:51:00.579062Z","submitted_at":"2026-03-30T02:23:37Z","title":"Towards Efficient Large Vision-Language Models: A Comprehensive Survey on Inference Strategies","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-14T21:40:55.343169Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2603.27960"},"observation_digest":"sha256:833035714585512d7722b97924d0f9ad1672983fe64260775e1e31993b5dbd11","observation_id":"da2b1e96-3bd8-47b1-81d1-29da03e930ce","resolution":{"observed_at":"2026-05-14T21:43:00.694286Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":"2406.17565","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-07-04T09:39:46.803577Z","title":"Memserve: Con- text caching for disaggregated llm serving with elastic memory pool.arXiv preprint arXiv:2406.17565","venue":null,"work_id":"51c7ec72-def3-42da-9410-c0152debc67f","year":2024},"citing_paper":{"arxiv_id":"2604.07173","last_updated":"2026-04-08T15:01:04Z","snapshot_observed_at":"2026-07-06T22:55:28.855291Z","submitted_at":"2026-04-08T15:01:04Z","title":"InfiniLoRA: Disaggregated Multi-LoRA Serving for Large Language Models","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T17:23:40.872418Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2604.07173"},"observation_digest":"sha256:6a3151ac60c3050493b7a8ba50d58354df7d6faba3f432739185722bb678ccdf","observation_id":"3101e9b5-02e8-46ba-8f2e-d1bca161f339","resolution":{"observed_at":"2026-05-11T06:55:59.356849Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":"2406.17565","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-07-04T09:39:46.803577Z","title":"Memserve: Con- text caching for disaggregated llm serving with elastic memory pool.arXiv preprint arXiv:2406.17565","venue":null,"work_id":"51c7ec72-def3-42da-9410-c0152debc67f","year":2024},"citing_paper":{"arxiv_id":"2605.02329","last_updated":"2026-05-25T17:26:33Z","snapshot_observed_at":"2026-08-02T00:39:56.914878Z","submitted_at":"2026-05-04T08:29:47Z","title":"Taming Request Imbalance: SLO-Aware Scheduling for Disaggregated LLM Inference","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-08T18:27:13.781231Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2605.02329"},"observation_digest":"sha256:637a8f7e57b76341a70118e55d91eb4aefdc81de68192aaf0959a07711484659","observation_id":"fd481473-3656-4800-8567-ddac1ac91067","resolution":{"observed_at":"2026-05-09T06:25:46.412499Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":"2406.17565","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-07-04T09:39:46.803577Z","title":"Memserve: Con- text caching for disaggregated llm serving with elastic memory pool.arXiv preprint arXiv:2406.17565","venue":null,"work_id":"51c7ec72-def3-42da-9410-c0152debc67f","year":2024},"citing_paper":{"arxiv_id":"2605.02329","last_updated":"2026-05-25T17:26:33Z","snapshot_observed_at":"2026-08-02T00:39:56.914878Z","submitted_at":"2026-05-04T08:29:47Z","title":"Taming Request Imbalance: SLO-Aware Scheduling for Disaggregated LLM Inference","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-01T00:31:12.965178Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2605.02329"},"observation_digest":"sha256:562930649dec1cae5f8095220696c64f7d9677765a5ea4d2b314b707ec10f742","observation_id":"aef8e12f-c4e3-4c25-9137-6f58e5ec7722","resolution":{"observed_at":"2026-07-01T00:35:10.226400Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":"2406.17565","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-07-04T09:39:46.803577Z","title":"Memserve: Con- text caching for disaggregated llm serving with elastic memory pool.arXiv preprint arXiv:2406.17565","venue":null,"work_id":"51c7ec72-def3-42da-9410-c0152debc67f","year":2024},"citing_paper":{"arxiv_id":"2605.22850","last_updated":"2026-05-16T16:48:46Z","snapshot_observed_at":"2026-08-02T10:46:09.855361Z","submitted_at":"2026-05-16T16:48:46Z","title":"ObjectCache: Layerwise Object-Storage Retrieval for KV Cache Reuse","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-25T00:22:05.867914Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2605.22850"},"observation_digest":"sha256:8c76f11a682bef51a23ff6017643fbe8a23d52ae8f5c1da799f219993b006a20","observation_id":"e463e356-0d2e-4706-94b3-2b2a0c63b35a","resolution":{"observed_at":"2026-05-25T00:25:08.358107Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":"2406.17565","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-07-04T09:39:46.803577Z","title":"Memserve: Con- text caching for disaggregated llm serving with elastic memory pool.arXiv preprint arXiv:2406.17565","venue":null,"work_id":"51c7ec72-def3-42da-9410-c0152debc67f","year":2024},"citing_paper":{"arxiv_id":"2605.23389","last_updated":"2026-05-22T09:00:45Z","snapshot_observed_at":"2026-08-01T23:40:06.222566Z","submitted_at":"2026-05-22T09:00:45Z","title":"AlignedServe: Orchestrating Prefix-aware Batching to Build a High-throughput and Computing-efficient LLM Serving System","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-25T03:12:49.028342Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2605.23389"},"observation_digest":"sha256:a9741d26a6b934be8b90d1c676ae2a4034438880de6388423e87c5494d61982c","observation_id":"e820ca28-003b-4c8e-a73c-6b4fa10c1ba3","resolution":{"observed_at":"2026-05-25T03:15:17.299404Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":"2406.17565","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-07-04T09:39:46.803577Z","title":"Memserve: Con- text caching for disaggregated llm serving with elastic memory pool.arXiv preprint arXiv:2406.17565","venue":null,"work_id":"51c7ec72-def3-42da-9410-c0152debc67f","year":2024},"citing_paper":{"arxiv_id":"2606.00866","last_updated":"2026-05-30T19:44:25Z","snapshot_observed_at":"2026-08-06T18:53:41.253747Z","submitted_at":"2026-05-30T19:44:25Z","title":"Idleness is Relative: Exploiting Tool-Call Idle Windows for Offloading in Agentic Systems with MORI","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-28T17:30:56.324289Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2606.00866"},"observation_digest":"sha256:f45ec0ed1a100a9d51fba6393309aa95aeed8d725e2b8f9c404a433bd29ca650","observation_id":"9c02aad4-b481-4ca1-877e-d54c2f24c980","resolution":{"observed_at":"2026-07-01T21:06:13.627198Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":"2406.17565","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-07-04T09:39:46.803577Z","title":"Memserve: Con- text caching for disaggregated llm serving with elastic memory pool.arXiv preprint arXiv:2406.17565","venue":null,"work_id":"51c7ec72-def3-42da-9410-c0152debc67f","year":2024},"citing_paper":{"arxiv_id":"2606.01065","last_updated":"2026-05-31T07:13:15Z","snapshot_observed_at":"2026-08-07T18:48:40.880591Z","submitted_at":"2026-05-31T07:13:15Z","title":"Leyline: KV Cache Directives for Agentic Inference","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-06-28T16:51:08.114092Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2606.01065"},"observation_digest":"sha256:f429d40d564187e501e1f8a343c8b99d75d1df7727bc13aef1ea69ccdc72a5fb","observation_id":"2fd66820-e86d-4091-940f-115992c883d0","resolution":{"observed_at":"2026-07-01T21:36:14.553901Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":"2406.17565","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-07-04T09:39:46.803577Z","title":"Memserve: Con- text caching for disaggregated llm serving with elastic memory pool.arXiv preprint arXiv:2406.17565","venue":null,"work_id":"51c7ec72-def3-42da-9410-c0152debc67f","year":2024},"citing_paper":{"arxiv_id":"2606.22541","last_updated":"2026-06-21T14:57:45Z","snapshot_observed_at":"2026-08-05T19:10:35.929740Z","submitted_at":"2026-06-21T14:57:45Z","title":"ASAP: A Disaggregated and Asynchronous Inference System for MoE Prefill","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-26T09:42:39.573568Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2606.22541"},"observation_digest":"sha256:dbeb769ce8c361f9e41f6e1d5ca9aaa515d3405abaee67d2e855cf5dd0393f34","observation_id":"116ba9ea-3c2a-4e3b-8b0d-22aa9632bda0","resolution":{"observed_at":"2026-07-04T09:39:46.805074Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":"2406.17565","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-07-04T09:39:46.803577Z","title":"Memserve: Con- text caching for disaggregated llm serving with elastic memory pool.arXiv preprint arXiv:2406.17565","venue":null,"work_id":"51c7ec72-def3-42da-9410-c0152debc67f","year":2024},"citing_paper":{"arxiv_id":"2606.29207","last_updated":"2026-06-28T05:16:17Z","snapshot_observed_at":"2026-08-05T18:19:52.152288Z","submitted_at":"2026-06-28T05:16:17Z","title":"KernelFlume: Elastic Core-Attention Scaling for Agentic Long-Context Decoding","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-30T02:54:11.777615Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2606.29207"},"observation_digest":"sha256:a6e7edab97afd26a4bcf20db049ad5e5a997bffd3bee00036aa2acb0dabab728","observation_id":"ef9aefc8-c8cb-4722-b9f5-3ab05fb142b0","resolution":{"observed_at":"2026-06-30T03:04:14.223576Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-07-12T09:50:23.266920Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.02574","last_updated":"2026-06-30T16:12:40Z","snapshot_observed_at":"2026-07-12T09:50:22.497230Z","submitted_at":"2026-06-30T16:12:40Z","title":"From Tensor Buffer to Distributed Memory Hierarchy: A Survey of KV Cache Management for LLM Serving","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-07-12T09:50:23.266920Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2607.02574"},"observation_digest":"sha256:4c7725a895b196eb0d55484dc4fbe7a2db978113b7190354a6d4fad460bc2c46","observation_id":"fd7edb7a-853e-4df5-93bf-7fe941c9f6ca","resolution":{"observed_at":"2026-07-12T09:50:23.266920Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-07-14T07:51:15.549754Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.10987","last_updated":"2026-07-13T01:19:13Z","snapshot_observed_at":"2026-08-07T15:37:46.507242Z","submitted_at":"2026-07-13T01:19:13Z","title":"[AAFLOW+] Stateful Operator Abstraction with Zero-Copy Distributed KV Cache Orchestration for Multi-Agent Workflows","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-07-14T07:51:15.549754Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2607.10987"},"observation_digest":"sha256:8f1022dc3d18e82ab105e7e474837293be06e6726527f6c939ec3cdde5aed62e","observation_id":"9bd133b5-8c0d-498d-af8e-8c50ce925c61","resolution":{"observed_at":"2026-07-14T07:51:15.549754Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-08-01T15:06:54.222623Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.26475","last_updated":"2026-07-29T05:01:42Z","snapshot_observed_at":"2026-08-07T15:06:08.198308Z","submitted_at":"2026-07-29T05:01:42Z","title":"DualDecoder: Accelerate Long Context LLM Inference by Predictive Prefetch","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-01T15:06:54.222623Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2607.26475"},"observation_digest":"sha256:da60e1d912da0ca44ab49d01e86878caf83b2075e74043efc553ee3e477842c4","observation_id":"98f1939f-4852-44c6-ad80-d2c249913367","resolution":{"observed_at":"2026-08-01T15:06:54.222623Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-08-03T14:14:10.290433Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29069","last_updated":"2026-07-31T06:44:06Z","snapshot_observed_at":"2026-08-07T03:44:55.840492Z","submitted_at":"2026-07-31T06:44:06Z","title":"Rethinking AI Cloud Infrastructure for Agentic Serving Systems with the Aries Experimentation Framework","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-03T14:14:10.290433Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2607.29069"},"observation_digest":"sha256:cc469be3d0949a3f5c004764af5b9f55b7b1b74623e531044eefdec34bc67239","observation_id":"c63fd332-4148-4190-a355-98bd631e8bc3","resolution":{"observed_at":"2026-08-03T14:14:10.290433Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-08-06T00:33:27.394771Z","title":"Memserve: Context caching for dis- aggregated LLM serving with elastic memory pool,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.01126","last_updated":"2026-08-02T09:51:15Z","snapshot_observed_at":"2026-08-09T00:37:13.692433Z","submitted_at":"2026-08-02T09:51:15Z","title":"Spatial Prefix Caching for Wireless Edge LLM Inference: A Stochastic-Geometry and Queueing Framework","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T00:33:27.394771Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2608.01126"},"observation_digest":"sha256:37b96e199d7ace553ea5457920429ce5951ad1f295c0cf887eac9cee9ce34993","observation_id":"251d395c-dede-4ccd-bd4c-523d8ccf8a65","resolution":{"observed_at":"2026-08-06T00:33:27.394771Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.17565","snapshot_observed_at":"2026-08-04T18:49:22.840559Z","title":"Memserve: Context caching for disaggregated llm serving with elastic memory pool,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.01891","last_updated":"2026-08-03T08:32:50Z","snapshot_observed_at":"2026-08-06T23:33:24.457506Z","submitted_at":"2026-08-03T08:32:50Z","title":"Energy-Efficient LLM Serving via Disaggregated Attention--FFN and Flexible Frequency Scaling","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-04T18:49:22.840559Z"},"links":{"cited_paper":"/paper/2406.17565","citing_paper":"/paper/2608.01891"},"observation_digest":"sha256:00e379c93ebc584806217b518564f24bc798546dac8dcc70c4c865296b03fd42","observation_id":"2b17282e-2cbb-4913-896d-4eaa09d08b5c","resolution":{"observed_at":"2026-08-04T18:49:22.840559Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2406.17565/citation-record","integrity":"/paper/2406.17565/integrity","json":"/paper/2406.17565/citation-record.json","paper":"/paper/2406.17565"},"outbound":[],"paper":{"arxiv_id":"2406.17565","last_updated":"2024-12-21T13:55:49Z","latest_version":3,"primary_category":"cs.DC","snapshot_observed_at":"2026-07-06T18:36:42.830927Z","submitted_at":"2024-06-25T14:02:08Z","title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 29 inbound Pith citation observations for arXiv:2406.17565."}