{"as_of":"2026-08-11T19:38:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:55461fcf41cb6745e31026d6214ef0540e8b6c715c924d19ca314d0266289c13","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":6,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":6,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":6,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":6,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T16:47:23.670684Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-21T15:30:17.994206Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2504.14966","last_updated":"2025-06-12T01:39:50Z","snapshot_observed_at":"2026-08-07T16:00:38.137237Z","submitted_at":"2025-04-21T08:48:48Z","title":"SLO-Aware Scheduling for Large Language Model Inferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.14966","snapshot_observed_at":"2026-08-09T16:47:23.670684Z","title":"Slo-aware scheduling for large language model inferences","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2502.01070","last_updated":"2025-08-25T06:01:25Z","snapshot_observed_at":"2026-08-09T20:48:49.113870Z","submitted_at":"2025-02-03T05:26:22Z","title":"An Inquiry into Datacenter TCO for LLM Inference with FP8","version":4},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-09T16:47:23.670684Z"},"links":{"cited_paper":"/paper/2504.14966","citing_paper":"/paper/2502.01070"},"observation_digest":"sha256:a169091a0d6efd69f32f110a69aed4f641009258c6fc06514526b33ea93ab583","observation_id":"a27c69f4-6c39-40a5-bf43-14c0544207b6","resolution":{"observed_at":"2026-08-09T16:47:23.670684Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.14966","last_updated":"2025-06-12T01:39:50Z","snapshot_observed_at":"2026-08-07T16:00:38.137237Z","submitted_at":"2025-04-21T08:48:48Z","title":"SLO-Aware Scheduling for Large Language Model Inferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.14966","snapshot_observed_at":"2026-08-05T16:21:09.138449Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.18736","last_updated":"2025-08-26T07:09:09Z","snapshot_observed_at":"2026-08-09T04:17:30.728713Z","submitted_at":"2025-08-26T07:09:09Z","title":"Rethinking Caching for LLM Serving Systems: Beyond Traditional Heuristics","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-05T16:21:09.138449Z"},"links":{"cited_paper":"/paper/2504.14966","citing_paper":"/paper/2508.18736"},"observation_digest":"sha256:8d807232d387455bb10bacce8c3269ca897a88d152418aa30c8b0a63a420e71e","observation_id":"9023d0cc-e6af-4fb3-8471-e632336756a3","resolution":{"observed_at":"2026-08-05T16:21:09.138449Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.14966","last_updated":"2025-06-12T01:39:50Z","snapshot_observed_at":"2026-08-07T16:00:38.137237Z","submitted_at":"2025-04-21T08:48:48Z","title":"SLO-Aware Scheduling for Large Language Model Inferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.14966","snapshot_observed_at":"2026-08-04T23:49:29.217377Z","title":"Slo-aware scheduling for large language model inferences,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.06362","last_updated":"2025-09-08T06:23:09Z","snapshot_observed_at":"2026-08-11T09:18:19.437706Z","submitted_at":"2025-09-08T06:23:09Z","title":"MaaSO: SLO-aware Orchestration of Heterogeneous Model Instances for MaaS","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-04T23:49:29.217377Z"},"links":{"cited_paper":"/paper/2504.14966","citing_paper":"/paper/2509.06362"},"observation_digest":"sha256:8d6ccf8cd18d119ff99028db273aa078afe629a0dc96b287abd084892b78f5c9","observation_id":"d9ed25c2-6a76-479b-b27f-ccf7755b806d","resolution":{"observed_at":"2026-08-04T23:49:29.217377Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.14966","last_updated":"2025-06-12T01:39:50Z","snapshot_observed_at":"2026-08-07T16:00:38.137237Z","submitted_at":"2025-04-21T08:48:48Z","title":"SLO-Aware Scheduling for Large Language Model Inferences","version":2},"cited_work":{"arxiv_id":"2504.14966","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.14966","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Slo-aware scheduling for large language model inferences.arXiv preprint arXiv:2504.14966","venue":null,"work_id":"c58c1089-7c66-4bd6-bfc9-41dd63ba5e08","year":2025},"citing_paper":{"arxiv_id":"2601.11652","last_updated":"2026-04-07T01:02:52Z","snapshot_observed_at":"2026-08-11T14:14:01.400167Z","submitted_at":"2026-01-15T16:46:01Z","title":"WISP: Waste- and Interference-Suppressed Distributed Speculative LLM Serving at the Edge via Dynamic Drafting and SLO-Aware Batching","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-16T14:12:06.034679Z"},"links":{"cited_paper":"/paper/2504.14966","citing_paper":"/paper/2601.11652"},"observation_digest":"sha256:f059f1f9df42e83ba55756d6126fc94016ea995cc59eecaa2254d6076035a064","observation_id":"4783fb87-3cd6-46d2-8866-cc03634ce6c8","resolution":{"observed_at":"2026-05-16T14:12:58.633365Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.14966","last_updated":"2025-06-12T01:39:50Z","snapshot_observed_at":"2026-08-07T16:00:38.137237Z","submitted_at":"2025-04-21T08:48:48Z","title":"SLO-Aware Scheduling for Large Language Model Inferences","version":2},"cited_work":{"arxiv_id":"2504.14966","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.14966","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Slo-aware scheduling for large language model inferences.arXiv preprint arXiv:2504.14966","venue":null,"work_id":"c58c1089-7c66-4bd6-bfc9-41dd63ba5e08","year":2025},"citing_paper":{"arxiv_id":"2601.20309","last_updated":"2026-05-18T19:51:16Z","snapshot_observed_at":"2026-07-06T22:43:17.026472Z","submitted_at":"2026-01-28T07:01:46Z","title":"SuperInfer: SLO-Aware Rotary Scheduling and Memory Management for LLM Inference on Superchips","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-21T15:26:01.283448Z"},"links":{"cited_paper":"/paper/2504.14966","citing_paper":"/paper/2601.20309"},"observation_digest":"sha256:610b7b35d1a9fc9406a3b49d21258339c4611cf142cb6bfcaf60f4bcbef1b59a","observation_id":"4de8fa78-d194-4e99-a21c-8949d2743483","resolution":{"observed_at":"2026-05-21T15:30:17.996464Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.14966","last_updated":"2025-06-12T01:39:50Z","snapshot_observed_at":"2026-08-07T16:00:38.137237Z","submitted_at":"2025-04-21T08:48:48Z","title":"SLO-Aware Scheduling for Large Language Model Inferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.14966","snapshot_observed_at":"2026-08-02T14:07:59.627222Z","title":"Slo-aware scheduling for large language model inferences, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.18253","last_updated":"2026-05-13T20:29:09Z","snapshot_observed_at":"2026-08-08T22:52:01.648545Z","submitted_at":"2026-05-13T20:29:09Z","title":"Beyond Accuracy and Cost: Latency-Aware LLM Query Routing for Dynamic Workloads","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-02T14:07:59.627222Z"},"links":{"cited_paper":"/paper/2504.14966","citing_paper":"/paper/2607.18253"},"observation_digest":"sha256:1843c9a38504c170c24b3b7068c13465bfa137d863d745e16f9c327de4adb294","observation_id":"03f5a96c-5519-45f6-9c5f-917449ee6274","resolution":{"observed_at":"2026-08-02T14:07:59.627222Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2504.14966/citation-record","integrity":"/paper/2504.14966/integrity","json":"/paper/2504.14966/citation-record.json","paper":"/paper/2504.14966"},"outbound":[],"paper":{"arxiv_id":"2504.14966","last_updated":"2025-06-12T01:39:50Z","latest_version":2,"primary_category":"cs.DC","snapshot_observed_at":"2026-08-07T16:00:38.137237Z","submitted_at":"2025-04-21T08:48:48Z","title":"SLO-Aware Scheduling for Large Language Model Inferences"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 6 inbound Pith citation observations for arXiv:2504.14966."}