{"as_of":"2026-08-10T20:32:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e19ebe24bcce949ad96fc4c4bea5bc41276b5005e87e6625f99cae64640291a9","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":20,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":20,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":20,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":20,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T22:01:23.711196Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T08:39:41.968159Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2404.08509","last_updated":"2024-11-25T17:35:07Z","snapshot_observed_at":"2026-08-09T09:17:59.811981Z","submitted_at":"2024-04-12T14:46:15Z","title":"Efficient Interactive LLM Serving with Proxy Model-based Sequence Length Prediction","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.08509","snapshot_observed_at":"2026-08-06T22:01:23.711196Z","title":"arXiv:2404.08509 [cs.DC] https://arxiv.org/abs/2404.08509","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.22950","last_updated":"2025-06-28T16:52:29Z","snapshot_observed_at":"2026-08-09T21:07:16.042677Z","submitted_at":"2025-06-28T16:52:29Z","title":"Infinite Sampling: Efficient and Stable Grouped RL Training for Large Language Models","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T22:01:23.711196Z"},"links":{"cited_paper":"/paper/2404.08509","citing_paper":"/paper/2506.22950"},"observation_digest":"sha256:e802f0840353c1cfbf08e4c7f1d762e9cf50f96baa21b9469a9c6b88be57ecf7","observation_id":"7831e371-0d79-4ee5-8c34-0a809bbc6b12","resolution":{"observed_at":"2026-08-06T22:01:23.711196Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08509","last_updated":"2024-11-25T17:35:07Z","snapshot_observed_at":"2026-08-09T09:17:59.811981Z","submitted_at":"2024-04-12T14:46:15Z","title":"Efficient Interactive LLM Serving with Proxy Model-based Sequence Length Prediction","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.08509","snapshot_observed_at":"2026-08-06T10:22:42.502527Z","title":"Efficient interactive LLM serving with proxy model-based sequence length prediction,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.502527Z"},"links":{"cited_paper":"/paper/2404.08509","citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:514f63e64cddfc5cc3e07a1883b5f8e1caacc34dc69affdf627a65cb2575d06b","observation_id":"6f8399dc-ff72-4234-bc1e-d302dff8fba4","resolution":{"observed_at":"2026-08-06T10:22:42.502527Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08509","last_updated":"2024-11-25T17:35:07Z","snapshot_observed_at":"2026-08-09T09:17:59.811981Z","submitted_at":"2024-04-12T14:46:15Z","title":"Efficient Interactive LLM Serving with Proxy Model-based Sequence Length Prediction","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.08509","snapshot_observed_at":"2026-08-04T13:29:01.810550Z","title":"T., BA ¸ SAR, T.,ANDIYER, R","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2510.00481","last_updated":"2026-05-30T15:21:07Z","snapshot_observed_at":"2026-08-08T11:19:05.561387Z","submitted_at":"2025-10-01T04:03:51Z","title":"Make a Video Call with LLM: A Measurement Campaign over Six Mainstream Apps","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-04T13:29:01.810550Z"},"links":{"cited_paper":"/paper/2404.08509","citing_paper":"/paper/2510.00481"},"observation_digest":"sha256:04eb6a6591314a7243ecb69004888c3ea7e1b3c4526156a853304f7fd506ab24","observation_id":"7f82b0bb-98e2-4a74-bc9e-00fef6d80ae5","resolution":{"observed_at":"2026-08-04T13:29:01.810550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08509","last_updated":"2024-11-25T17:35:07Z","snapshot_observed_at":"2026-08-09T09:17:59.811981Z","submitted_at":"2024-04-12T14:46:15Z","title":"Efficient Interactive LLM Serving with Proxy Model-based Sequence Length Prediction","version":2},"cited_work":{"arxiv_id":"2404.08509","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.08509","snapshot_observed_at":"2026-07-04T08:39:41.968159Z","title":"Efficient interactive llm serving with proxy model-based sequence length prediction","venue":null,"work_id":"3f3bde8b-bcc8-4d4a-8eed-61dceb8e2df1","year":2024},"citing_paper":{"arxiv_id":"2510.13668","last_updated":"2026-05-04T10:30:41Z","snapshot_observed_at":"2026-08-08T19:04:42.116042Z","submitted_at":"2025-10-15T15:29:08Z","title":"STAR: Decode-Phase Rescheduling for LLM Inference","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-18T06:11:28.120860Z"},"links":{"cited_paper":"/paper/2404.08509","citing_paper":"/paper/2510.13668"},"observation_digest":"sha256:d2af8896b0477c73fb6bc02726af69eee6245be21d34ead690a61707c481620d","observation_id":"5d527a7b-b297-408f-9e99-36cf7edf74f4","resolution":{"observed_at":"2026-05-18T06:12:25.932914Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08509","last_updated":"2024-11-25T17:35:07Z","snapshot_observed_at":"2026-08-09T09:17:59.811981Z","submitted_at":"2024-04-12T14:46:15Z","title":"Efficient Interactive LLM Serving with Proxy Model-based Sequence Length Prediction","version":2},"cited_work":{"arxiv_id":"2404.08509","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.08509","snapshot_observed_at":"2026-07-04T08:39:41.968159Z","title":"Efficient interactive llm serving with proxy model-based sequence length prediction","venue":null,"work_id":"3f3bde8b-bcc8-4d4a-8eed-61dceb8e2df1","year":2024},"citing_paper":{"arxiv_id":"2512.19179","last_updated":"2026-05-15T11:32:47Z","snapshot_observed_at":"2026-08-03T00:09:02.053747Z","submitted_at":"2025-12-22T09:13:40Z","title":"CascadeInfer: Length-Aware Scheduling of LLM Serving with Low Latency and Load Balancing","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-21T17:24:59.556926Z"},"links":{"cited_paper":"/paper/2404.08509","citing_paper":"/paper/2512.19179"},"observation_digest":"sha256:6184115ae5081732bc973d8f9555e873714b5e204caa5d8380dd6bac2cea9f9b","observation_id":"5bf92d1d-3317-4373-90da-738cec23475a","resolution":{"observed_at":"2026-05-21T17:25:25.275238Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08509","last_updated":"2024-11-25T17:35:07Z","snapshot_observed_at":"2026-08-09T09:17:59.811981Z","submitted_at":"2024-04-12T14:46:15Z","title":"Efficient Interactive LLM Serving with Proxy Model-based Sequence Length Prediction","version":2},"cited_work":{"arxiv_id":"2404.08509","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.08509","snapshot_observed_at":"2026-07-04T08:39:41.968159Z","title":"Efficient interactive llm serving with proxy model-based sequence length prediction","venue":null,"work_id":"3f3bde8b-bcc8-4d4a-8eed-61dceb8e2df1","year":2024},"citing_paper":{"arxiv_id":"2601.20309","last_updated":"2026-05-18T19:51:16Z","snapshot_observed_at":"2026-07-06T22:43:17.026472Z","submitted_at":"2026-01-28T07:01:46Z","title":"SuperInfer: SLO-Aware Rotary Scheduling and Memory Management for LLM Inference on Superchips","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-21T15:26:01.283448Z"},"links":{"cited_paper":"/paper/2404.08509","citing_paper":"/paper/2601.20309"},"observation_digest":"sha256:1cca06aed2edc50b7d0532d820af208e8c6459e306ae35622b3ea2d498ea9273","observation_id":"b887a3d0-c840-4cb7-8a41-1053673f8a5d","resolution":{"observed_at":"2026-05-21T15:30:17.984441Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08509","last_updated":"2024-11-25T17:35:07Z","snapshot_observed_at":"2026-08-09T09:17:59.811981Z","submitted_at":"2024-04-12T14:46:15Z","title":"Efficient Interactive LLM Serving with Proxy Model-based Sequence Length Prediction","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.08509","snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"cited_paper":"/paper/2404.08509","citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:d0c1f5016e6b819327f527b314b32b75283cb57e05dc5f7a43893dd70f971e5f","observation_id":"480d42c1-e30b-4030-9696-52950dec518b","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08509","last_updated":"2024-11-25T17:35:07Z","snapshot_observed_at":"2026-08-09T09:17:59.811981Z","submitted_at":"2024-04-12T14:46:15Z","title":"Efficient Interactive LLM Serving with Proxy Model-based Sequence Length Prediction","version":2},"cited_work":{"arxiv_id":"2404.08509","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.08509","snapshot_observed_at":"2026-07-04T08:39:41.968159Z","title":"Efficient interactive llm serving with proxy model-based sequence length prediction","venue":null,"work_id":"3f3bde8b-bcc8-4d4a-8eed-61dceb8e2df1","year":2024},"citing_paper":{"arxiv_id":"2604.07144","last_updated":"2026-04-08T14:37:17Z","snapshot_observed_at":"2026-07-06T22:55:28.855291Z","submitted_at":"2026-04-08T14:37:17Z","title":"Autopoiesis: A Self-Evolving System Paradigm for LLM Serving Under Runtime Dynamics","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-10T17:30:40.021376Z"},"links":{"cited_paper":"/paper/2404.08509","citing_paper":"/paper/2604.07144"},"observation_digest":"sha256:17dd589a9a370f83f35a5bb50d84802bccccba825b94a7dbf3994cd4691ed798","observation_id":"6e2d4b04-a066-4c30-8278-ae2f3481830f","resolution":{"observed_at":"2026-05-11T06:41:19.816284Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08509","last_updated":"2024-11-25T17:35:07Z","snapshot_observed_at":"2026-08-09T09:17:59.811981Z","submitted_at":"2024-04-12T14:46:15Z","title":"Efficient Interactive LLM Serving with Proxy Model-based Sequence Length Prediction","version":2},"cited_work":{"arxiv_id":"2404.08509","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.08509","snapshot_observed_at":"2026-07-04T08:39:41.968159Z","title":"Efficient interactive llm serving with proxy model-based sequence length prediction","venue":null,"work_id":"3f3bde8b-bcc8-4d4a-8eed-61dceb8e2df1","year":2024},"citing_paper":{"arxiv_id":"2604.07931","last_updated":"2026-04-09T07:49:52Z","snapshot_observed_at":"2026-08-01T20:01:12.346839Z","submitted_at":"2026-04-09T07:49:52Z","title":"Robust Length Prediction: A Perspective from Heavy-Tailed Prompt-Conditioned Distributions","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-05-10T18:32:42.905852Z"},"links":{"cited_paper":"/paper/2404.08509","citing_paper":"/paper/2604.07931"},"observation_digest":"sha256:83fba912175b2ab4d38dfc2cd583670e5e15e1a2438c70e0c08b4eada485cbc1","observation_id":"194f3840-2a3d-4c94-acab-8328da2be2c4","resolution":{"observed_at":"2026-05-11T00:25:50.663702Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08509","last_updated":"2024-11-25T17:35:07Z","snapshot_observed_at":"2026-08-09T09:17:59.811981Z","submitted_at":"2024-04-12T14:46:15Z","title":"Efficient Interactive LLM Serving with Proxy Model-based Sequence Length Prediction","version":2},"cited_work":{"arxiv_id":"2404.08509","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.08509","snapshot_observed_at":"2026-07-04T08:39:41.968159Z","title":"Efficient interactive llm serving with proxy model-based sequence length prediction","venue":null,"work_id":"3f3bde8b-bcc8-4d4a-8eed-61dceb8e2df1","year":2024},"citing_paper":{"arxiv_id":"2605.04595","last_updated":"2026-05-06T07:42:26Z","snapshot_observed_at":"2026-08-08T18:04:13.922410Z","submitted_at":"2026-05-06T07:42:26Z","title":"A Queueing-Theoretic Framework for Stability Analysis of LLM Inference with KV Cache Memory Constraints","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-05-08T17:28:05.492863Z"},"links":{"cited_paper":"/paper/2404.08509","citing_paper":"/paper/2605.04595"},"observation_digest":"sha256:df75bc049f3e7218516ba78787a6aa44426d97164d7d213937ca056827ea7096","observation_id":"ebf300e4-9ae2-4898-bf40-9b833550ff0d","resolution":{"observed_at":"2026-05-11T17:31:08.962305Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08509","last_updated":"2024-11-25T17:35:07Z","snapshot_observed_at":"2026-08-09T09:17:59.811981Z","submitted_at":"2024-04-12T14:46:15Z","title":"Efficient Interactive LLM Serving with Proxy Model-based Sequence Length Prediction","version":2},"cited_work":{"arxiv_id":"2404.08509","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.08509","snapshot_observed_at":"2026-07-04T08:39:41.968159Z","title":"Efficient interactive llm serving with proxy model-based sequence length prediction","venue":null,"work_id":"3f3bde8b-bcc8-4d4a-8eed-61dceb8e2df1","year":2024},"citing_paper":{"arxiv_id":"2605.06113","last_updated":"2026-05-08T02:53:17Z","snapshot_observed_at":"2026-07-06T23:18:36.537416Z","submitted_at":"2026-05-07T12:25:38Z","title":"Tackling the Data-Parallel Load Balancing Bottleneck in LLM Serving: Practical Online Routing at Scale","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-05-08T05:22:14.122156Z"},"links":{"cited_paper":"/paper/2404.08509","citing_paper":"/paper/2605.06113"},"observation_digest":"sha256:374c837badad60df4ad3e8386345c9b6c52742c374a368c3abfc667893e2d022","observation_id":"087b1463-2ff7-4def-a36b-49e8544a58a0","resolution":{"observed_at":"2026-05-11T21:31:14.071579Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08509","last_updated":"2024-11-25T17:35:07Z","snapshot_observed_at":"2026-08-09T09:17:59.811981Z","submitted_at":"2024-04-12T14:46:15Z","title":"Efficient Interactive LLM Serving with Proxy Model-based Sequence Length Prediction","version":2},"cited_work":{"arxiv_id":"2404.08509","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.08509","snapshot_observed_at":"2026-07-04T08:39:41.968159Z","title":"Efficient interactive llm serving with proxy model-based sequence length prediction","venue":null,"work_id":"3f3bde8b-bcc8-4d4a-8eed-61dceb8e2df1","year":2024},"citing_paper":{"arxiv_id":"2605.06113","last_updated":"2026-05-08T02:53:17Z","snapshot_observed_at":"2026-07-06T23:18:36.537416Z","submitted_at":"2026-05-07T12:25:38Z","title":"Tackling the Data-Parallel Load Balancing Bottleneck in LLM Serving: Practical Online Routing at Scale","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-05-11T00:59:57.639525Z"},"links":{"cited_paper":"/paper/2404.08509","citing_paper":"/paper/2605.06113"},"observation_digest":"sha256:897a4f3cca77a4e4e88cab8b6442ce2931f76685b6f9ae845c32d9c25dbdf996","observation_id":"cedacd78-8a15-45a4-937f-9b3b6698bff6","resolution":{"observed_at":"2026-05-11T04:55:56.905374Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08509","last_updated":"2024-11-25T17:35:07Z","snapshot_observed_at":"2026-08-09T09:17:59.811981Z","submitted_at":"2024-04-12T14:46:15Z","title":"Efficient Interactive LLM Serving with Proxy Model-based Sequence Length Prediction","version":2},"cited_work":{"arxiv_id":"2404.08509","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.08509","snapshot_observed_at":"2026-07-04T08:39:41.968159Z","title":"Efficient interactive llm serving with proxy model-based sequence length prediction","venue":null,"work_id":"3f3bde8b-bcc8-4d4a-8eed-61dceb8e2df1","year":2024},"citing_paper":{"arxiv_id":"2606.07248","last_updated":"2026-07-30T08:24:42Z","snapshot_observed_at":"2026-08-08T08:55:43.105158Z","submitted_at":"2026-06-05T13:19:05Z","title":"Clairvoyant: Predictive Shortest-Job-First Admission for Serial LLM Inference","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-27T20:59:02.818223Z"},"links":{"cited_paper":"/paper/2404.08509","citing_paper":"/paper/2606.07248"},"observation_digest":"sha256:b468f1dbf135b9519753f7404f3f3eba12aa6b2954309ccfdefe9b5d82978d2d","observation_id":"2d07ea33-387c-4801-b984-d4ca49b5a361","resolution":{"observed_at":"2026-07-02T20:07:21.791431Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08509","last_updated":"2024-11-25T17:35:07Z","snapshot_observed_at":"2026-08-09T09:17:59.811981Z","submitted_at":"2024-04-12T14:46:15Z","title":"Efficient Interactive LLM Serving with Proxy Model-based Sequence Length Prediction","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.08509","snapshot_observed_at":"2026-08-02T12:16:31.347364Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.07248","last_updated":"2026-07-30T08:24:42Z","snapshot_observed_at":"2026-08-08T08:55:43.105158Z","submitted_at":"2026-06-05T13:19:05Z","title":"Clairvoyant: Predictive Shortest-Job-First Admission for Serial LLM Inference","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-02T12:16:31.347364Z"},"links":{"cited_paper":"/paper/2404.08509","citing_paper":"/paper/2606.07248"},"observation_digest":"sha256:3feeea2a2eafce4c5c58fcde1a8691c89be1e3f92047cc7be8cf7e23ab84ff3b","observation_id":"7212223b-05ec-4998-891a-2444fca7ee67","resolution":{"observed_at":"2026-08-02T12:16:31.347364Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08509","last_updated":"2024-11-25T17:35:07Z","snapshot_observed_at":"2026-08-09T09:17:59.811981Z","submitted_at":"2024-04-12T14:46:15Z","title":"Efficient Interactive LLM Serving with Proxy Model-based Sequence Length Prediction","version":2},"cited_work":{"arxiv_id":"2404.08509","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.08509","snapshot_observed_at":"2026-07-04T08:39:41.968159Z","title":"Efficient interactive llm serving with proxy model-based sequence length prediction","venue":null,"work_id":"3f3bde8b-bcc8-4d4a-8eed-61dceb8e2df1","year":2024},"citing_paper":{"arxiv_id":"2606.18431","last_updated":"2026-06-16T19:25:37Z","snapshot_observed_at":"2026-08-04T00:16:31.977200Z","submitted_at":"2026-06-16T19:25:37Z","title":"Beyond Prediction: Tail-Aware Scheduling for LLM Inference","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-06-27T01:01:34.655458Z"},"links":{"cited_paper":"/paper/2404.08509","citing_paper":"/paper/2606.18431"},"observation_digest":"sha256:a0efcd2ca7df16b4b7b900b089bb2b14475d26098357d563b9f16bf9d1379bc6","observation_id":"0b8bedb0-2a8d-408a-97ba-dfba847d6ee1","resolution":{"observed_at":"2026-07-03T20:58:57.865939Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08509","last_updated":"2024-11-25T17:35:07Z","snapshot_observed_at":"2026-08-09T09:17:59.811981Z","submitted_at":"2024-04-12T14:46:15Z","title":"Efficient Interactive LLM Serving with Proxy Model-based Sequence Length Prediction","version":2},"cited_work":{"arxiv_id":"2404.08509","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.08509","snapshot_observed_at":"2026-07-04T08:39:41.968159Z","title":"Efficient interactive llm serving with proxy model-based sequence length prediction","venue":null,"work_id":"3f3bde8b-bcc8-4d4a-8eed-61dceb8e2df1","year":2024},"citing_paper":{"arxiv_id":"2606.22327","last_updated":"2026-06-24T14:02:53Z","snapshot_observed_at":"2026-08-07T17:40:04.855669Z","submitted_at":"2026-06-21T04:05:38Z","title":"Geometry-Aware Online Scheduling for LLM Serving: From Theoretical Bound to System Practice","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-26T11:17:35.860286Z"},"links":{"cited_paper":"/paper/2404.08509","citing_paper":"/paper/2606.22327"},"observation_digest":"sha256:c3ba70c0e8c25ecf44935ab648d53860147f398b031e3328c2e2d3287b396623","observation_id":"98c7d8f5-f289-4f2c-902f-337f9da42e53","resolution":{"observed_at":"2026-07-04T08:39:41.969878Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08509","last_updated":"2024-11-25T17:35:07Z","snapshot_observed_at":"2026-08-09T09:17:59.811981Z","submitted_at":"2024-04-12T14:46:15Z","title":"Efficient Interactive LLM Serving with Proxy Model-based Sequence Length Prediction","version":2},"cited_work":{"arxiv_id":"2404.08509","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.08509","snapshot_observed_at":"2026-07-04T08:39:41.968159Z","title":"Efficient interactive llm serving with proxy model-based sequence length prediction","venue":null,"work_id":"3f3bde8b-bcc8-4d4a-8eed-61dceb8e2df1","year":2024},"citing_paper":{"arxiv_id":"2606.30391","last_updated":"2026-06-29T14:44:24Z","snapshot_observed_at":"2026-08-04T10:56:44.243323Z","submitted_at":"2026-06-29T14:44:24Z","title":"Energy-Aware Scheduling for Serverless LLM Serving on Shared GPUs","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-06-30T03:41:51.034169Z"},"links":{"cited_paper":"/paper/2404.08509","citing_paper":"/paper/2606.30391"},"observation_digest":"sha256:0c1501ac953484935f32753e25c243c4e6f0f8784e6cb5a4379b7922ad7c2ea9","observation_id":"283d4440-2c2a-4bf6-8fd9-aeb4d21e7e00","resolution":{"observed_at":"2026-07-01T15:15:48.671384Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08509","last_updated":"2024-11-25T17:35:07Z","snapshot_observed_at":"2026-08-09T09:17:59.811981Z","submitted_at":"2024-04-12T14:46:15Z","title":"Efficient Interactive LLM Serving with Proxy Model-based Sequence Length Prediction","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.08509","snapshot_observed_at":"2026-07-11T22:49:10.064714Z","title":"arXiv preprint arXiv:2404.08509 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.03948","last_updated":"2026-07-04T16:48:37Z","snapshot_observed_at":"2026-08-07T06:39:55.842387Z","submitted_at":"2026-07-04T16:48:37Z","title":"Online Linear Programming for Multi-Objective Routing in LLM Serving","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-07-11T22:49:10.064714Z"},"links":{"cited_paper":"/paper/2404.08509","citing_paper":"/paper/2607.03948"},"observation_digest":"sha256:0906c414c94ed2281f795fcedcf6715eed926119a8ee2a8c61653db449eeb606","observation_id":"144f9d3b-2a0f-4c9f-ae40-97a4e2c6276b","resolution":{"observed_at":"2026-07-11T22:49:10.064714Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08509","last_updated":"2024-11-25T17:35:07Z","snapshot_observed_at":"2026-08-09T09:17:59.811981Z","submitted_at":"2024-04-12T14:46:15Z","title":"Efficient Interactive LLM Serving with Proxy Model-based Sequence Length Prediction","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.08509","snapshot_observed_at":"2026-07-13T04:25:48.447055Z","title":"Kalbarczyk, Tamer Basar, and Ravishankar K","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.09248","last_updated":"2026-07-10T09:52:11Z","snapshot_observed_at":"2026-08-08T07:42:04.363588Z","submitted_at":"2026-07-10T09:52:11Z","title":"General Non-Clairvoyant KV-Cache Scheduling via Regime-Aware Routing","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-07-13T04:25:48.447055Z"},"links":{"cited_paper":"/paper/2404.08509","citing_paper":"/paper/2607.09248"},"observation_digest":"sha256:c021ae54c688ad84e9a19c113b83bcd753afb96ce90cea4d3bf125b869a2c20c","observation_id":"6074f733-8b52-4ad7-8e45-7b659fd43a4f","resolution":{"observed_at":"2026-07-13T04:25:48.447055Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08509","last_updated":"2024-11-25T17:35:07Z","snapshot_observed_at":"2026-08-09T09:17:59.811981Z","submitted_at":"2024-04-12T14:46:15Z","title":"Efficient Interactive LLM Serving with Proxy Model-based Sequence Length Prediction","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.08509","snapshot_observed_at":"2026-08-01T20:54:15.631451Z","title":"Efficient interactive llm serving with proxy model-based sequence length prediction,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.16488","last_updated":"2026-07-17T20:27:26Z","snapshot_observed_at":"2026-08-07T14:52:50.399447Z","submitted_at":"2026-07-17T20:27:26Z","title":"Auto-Scaling Heterogeneous Neural Processing Units for Energy and Cost-Efficient LLM Serving","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-01T20:54:15.631451Z"},"links":{"cited_paper":"/paper/2404.08509","citing_paper":"/paper/2607.16488"},"observation_digest":"sha256:c380dbedd94341d780c57624ccdec295c0685131e7543208341d346d81773144","observation_id":"42c23f12-c4f6-477f-81c8-b3cba8af8082","resolution":{"observed_at":"2026-08-01T20:54:15.631451Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2404.08509/citation-record","integrity":"/paper/2404.08509/integrity","json":"/paper/2404.08509/citation-record.json","paper":"/paper/2404.08509"},"outbound":[],"paper":{"arxiv_id":"2404.08509","last_updated":"2024-11-25T17:35:07Z","latest_version":2,"primary_category":"cs.DC","snapshot_observed_at":"2026-08-09T09:17:59.811981Z","submitted_at":"2024-04-12T14:46:15Z","title":"Efficient Interactive LLM Serving with Proxy Model-based Sequence Length Prediction"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 20 inbound Pith citation observations for arXiv:2404.08509."}