{"as_of":"2026-08-14T09:23:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6fd7fd1d0af3d57011d00e614388ac2f2eedd74f3357b0dfa1e522e1cb76520f","coverage":[{"denominator":12,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":12,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T18:11:16.388804Z","state":"measured"},{"denominator":17,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":17,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":5,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":5,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-03T15:10:12.647037Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-06-28T23:52:49.103329Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-13T20:44:23.894752Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.09019","snapshot_observed_at":"2026-08-03T15:10:12.647037Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.18020","last_updated":"2025-12-19T19:24:56Z","snapshot_observed_at":"2026-08-10T10:35:09.161054Z","submitted_at":"2025-12-19T19:24:56Z","title":"Specification and Detection of LLM Code Smells","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-03T15:10:12.647037Z"},"links":{"cited_paper":"/paper/2507.09019","citing_paper":"/paper/2512.18020"},"observation_digest":"sha256:b2f9d32bb1b7978e4597b0018a58e4289bafa13b19fdf89d72c6b413982e35b5","observation_id":"d3452ba6-8358-48f7-8b0e-968225f2155d","resolution":{"observed_at":"2026-08-03T15:10:12.647037Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-13T20:44:23.894752Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"cited_work":{"arxiv_id":"2507.09019","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.09019","snapshot_observed_at":"2026-06-28T23:52:49.103329Z","title":"On evaluating performance of llm inference serving systems","venue":null,"work_id":"16b1fb87-c364-4976-acde-225b20ac1e35","year":2025},"citing_paper":{"arxiv_id":"2602.01785","last_updated":"2026-04-28T16:05:53Z","snapshot_observed_at":"2026-07-06T22:44:04.951815Z","submitted_at":"2026-02-02T08:10:21Z","title":"CodeOCR: On the Effectiveness of Vision Language Models in Code Understanding","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-16T08:30:50.984873Z"},"links":{"cited_paper":"/paper/2507.09019","citing_paper":"/paper/2602.01785"},"observation_digest":"sha256:5adcc947ac33583adb61495f946ec34cdc38480613a3070a890faaaf7f9aec27","observation_id":"e853f491-9ddf-45e8-957b-e4d986b6e523","resolution":{"observed_at":"2026-05-16T08:32:36.437742Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-13T20:44:23.894752Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"cited_work":{"arxiv_id":"2507.09019","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.09019","snapshot_observed_at":"2026-06-28T23:52:49.103329Z","title":"On evaluating performance of llm inference serving systems","venue":null,"work_id":"16b1fb87-c364-4976-acde-225b20ac1e35","year":2025},"citing_paper":{"arxiv_id":"2604.09611","last_updated":"2026-03-12T10:10:37Z","snapshot_observed_at":"2026-08-14T01:58:08.479721Z","submitted_at":"2026-03-12T10:10:37Z","title":"Characterizing Performance-Energy Trade-offs of Large Language Models in Multi-Request Workflows","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-15T12:28:38.809950Z"},"links":{"cited_paper":"/paper/2507.09019","citing_paper":"/paper/2604.09611"},"observation_digest":"sha256:ed6ec5c6d24531328ea4181ed81dd8519176e5a311dba475bc6b83281bef34ad","observation_id":"9ad232a9-fc3f-4fd8-b1f5-c2238c852d2a","resolution":{"observed_at":"2026-05-15T12:30:00.285701Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-13T20:44:23.894752Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"cited_work":{"arxiv_id":"2507.09019","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.09019","snapshot_observed_at":"2026-06-28T23:52:49.103329Z","title":"On evaluating performance of llm inference serving systems","venue":null,"work_id":"16b1fb87-c364-4976-acde-225b20ac1e35","year":2025},"citing_paper":{"arxiv_id":"2604.15583","last_updated":"2026-04-24T15:54:27Z","snapshot_observed_at":"2026-08-10T21:44:14.791442Z","submitted_at":"2026-04-16T23:34:51Z","title":"SAGE: Selective Attention-Guided Extraction for Token-Efficient Document Indexing","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T08:32:02.222528Z"},"links":{"cited_paper":"/paper/2507.09019","citing_paper":"/paper/2604.15583"},"observation_digest":"sha256:d484162f1b14eb5145c1a9e8949b465fa2f4d233d31be8d916c0cde1bdd22bf6","observation_id":"8598cbcc-8abd-43f3-973d-68d1c2d75d7f","resolution":{"observed_at":"2026-05-10T08:32:52.126778Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-13T20:44:23.894752Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"cited_work":{"arxiv_id":"2507.09019","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.09019","snapshot_observed_at":"2026-06-28T23:52:49.103329Z","title":"On evaluating performance of llm inference serving systems","venue":null,"work_id":"16b1fb87-c364-4976-acde-225b20ac1e35","year":2025},"citing_paper":{"arxiv_id":"2605.29639","last_updated":"2026-05-28T09:07:06Z","snapshot_observed_at":"2026-08-14T08:47:30.574059Z","submitted_at":"2026-05-28T09:07:06Z","title":"RTP-LLM: High-Performance Alibaba LLM Inference Engine","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-28T23:52:40.763228Z"},"links":{"cited_paper":"/paper/2507.09019","citing_paper":"/paper/2605.29639"},"observation_digest":"sha256:98c5251625265d8cca430a6b5ef7ed9b4ad04d561cbc6ce7cedf3e4a31505725","observation_id":"9f3aa297-b3ea-47f7-8508-e3d155580047","resolution":{"observed_at":"2026-06-28T23:52:49.104862Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.09019/citation-record","integrity":"/paper/2507.09019/integrity","json":"/paper/2507.09019/citation-record.json","paper":"/paper/2507.09019"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2404.02015","last_updated":"2024-06-13T02:53:29Z","snapshot_observed_at":"2026-08-13T00:39:13.375770Z","submitted_at":"2024-04-02T14:56:43Z","title":"MuxServe: Flexible Spatial-Temporal Multiplexing for Multiple LLM Serving","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.02015","snapshot_observed_at":"2026-08-06T18:11:15.829163Z","title":"Muxserve: Flexible multiplexing for efficient multiple llm serving","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-13T20:44:23.894752Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T18:11:15.829163Z"},"links":{"cited_paper":"/paper/2404.02015","citing_paper":"/paper/2507.09019"},"observation_digest":"sha256:aa74692897e34226879f5cad822c40930701b3d48cf1e017c80a78186b9675a2","observation_id":"28100343-384c-4f3d-b8fb-69ae1e05919f","resolution":{"observed_at":"2026-08-06T18:11:15.829163Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.07240","last_updated":"2024-07-19T21:04:14Z","snapshot_observed_at":"2026-08-13T05:52:26.338685Z","submitted_at":"2023-10-11T07:08:20Z","title":"CacheGen: KV Cache Compression and Streaming for Fast Large Language Model Serving","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.07240","snapshot_observed_at":"2026-08-06T18:11:15.906010Z","title":"Cachegen: Fast context loading for language model applications","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-13T20:44:23.894752Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T18:11:15.906010Z"},"links":{"cited_paper":"/paper/2310.07240","citing_paper":"/paper/2507.09019"},"observation_digest":"sha256:4bd953403080e275245d2b4a5031f566c3a5a18dad49e862dfdd826fbbfca474","observation_id":"aa5dc71d-861b-48c0-9a3a-05f55e5e1dc4","resolution":{"observed_at":"2026-08-06T18:11:15.906010Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-08-12T23:36:32.131457Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-06T18:11:15.969489Z","title":"Noam Shazeer","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-13T20:44:23.894752Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T18:11:15.969489Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2507.09019"},"observation_digest":"sha256:274382064016a635f3e326fd5c8bc800a6ad2e933492c1568972d494ced68aee","observation_id":"9ebb2090-66ea-44ff-9cb0-72c5a1ba1f84","resolution":{"observed_at":"2026-08-06T18:11:15.969489Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:11:16.246971Z","title":"Guan Wang, Sijie Cheng, Xianyuan Zhan, Xiangang Li, Sen Song, and Yang Liu","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-13T20:44:23.894752Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T18:11:16.246971Z"},"links":{"citing_paper":"/paper/2507.09019"},"observation_digest":"sha256:2b23606f1f780d7491a58deeffdb5aea27f523aa4f335a2bf7c360f4b92dd442","observation_id":"dde8cc21-f904-417b-b360-f882410d750e","resolution":{"observed_at":"2026-08-06T18:11:16.246971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09526","last_updated":"2024-10-29T13:04:42Z","snapshot_observed_at":"2026-08-13T00:30:12.364886Z","submitted_at":"2024-04-15T07:45:04Z","title":"LoongServe: Efficiently Serving Long-Context Large Language Models with Elastic Sequence Parallelism","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09526","snapshot_observed_at":"2026-08-06T18:11:16.254256Z","title":"Loongserve: Efficiently serving long-context large language models with elastic sequence parallelism","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-13T20:44:23.894752Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T18:11:16.254256Z"},"links":{"cited_paper":"/paper/2404.09526","citing_paper":"/paper/2507.09019"},"observation_digest":"sha256:89799cd20bf7092306b18ea78c00cc30f11d14f2b4b7084bf33b4e8dbcb21c76","observation_id":"9182bd48-f2ae-424d-a307-cd01d9864f44","resolution":{"observed_at":"2026-08-06T18:11:16.254256Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12757","last_updated":"2025-05-25T14:08:01Z","snapshot_observed_at":"2026-08-12T22:58:57.541339Z","submitted_at":"2024-08-22T23:00:40Z","title":"NanoFlow: Towards Optimal Large Language Model Serving Throughput","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12757","snapshot_observed_at":"2026-08-06T18:11:16.262684Z","title":"Nanoflow: Towards optimal large language model serving throughput","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-13T20:44:23.894752Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T18:11:16.262684Z"},"links":{"cited_paper":"/paper/2408.12757","citing_paper":"/paper/2507.09019"},"observation_digest":"sha256:64f1851d0475fce0ef681822c4a5447db3b4f5742888a1474959e4da9ffd46dc","observation_id":"11274d95-91cf-475e-b1a9-373b85f355a5","resolution":{"observed_at":"2026-08-06T18:11:16.262684Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12757","last_updated":"2025-05-25T14:08:01Z","snapshot_observed_at":"2026-08-12T22:58:57.541339Z","submitted_at":"2024-08-22T23:00:40Z","title":"NanoFlow: Towards Optimal Large Language Model Serving Throughput","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12757","snapshot_observed_at":"2026-08-06T18:11:16.322242Z","title":"URL https://doi.org/10.48550/arXiv.2408.12757","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-13T20:44:23.894752Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T18:11:16.322242Z"},"links":{"cited_paper":"/paper/2408.12757","citing_paper":"/paper/2507.09019"},"observation_digest":"sha256:5b092ca12cead3d4bb7ab490e01cb59ceba8cdace1597d74cc0572bbb9666c23","observation_id":"55532324-5a12-45f1-b6fd-75421016dbd9","resolution":{"observed_at":"2026-08-06T18:11:16.322242Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:11:16.685457Z","title":"Parameter T uning (✓) In our judgment, the baselines don’t need parameter tuning","venue":null,"work_id":"dc1970ae-551f-4d54-8da7-7bec9d5b57bc","year":2023},"citing_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-13T20:44:23.894752Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T18:11:16.388804Z"},"links":{"citing_paper":"/paper/2507.09019"},"observation_digest":"sha256:271b2ed87b5e95f2470cc46829f20dc0cc8b0e21e69dc6a8beee8e104e1a2690","observation_id":"a9a0e147-24c3-4da7-9214-a9280e669f1b","resolution":{"observed_at":"2026-08-06T18:11:16.740082Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.03285","last_updated":"2024-06-05T06:06:43Z","snapshot_observed_at":"2026-08-13T05:31:57.731482Z","submitted_at":"2023-11-06T17:26:17Z","title":"S-LoRA: Serving Thousands of Concurrent LoRA Adapters","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.03285","snapshot_observed_at":"2026-08-06T18:11:16.146190Z","title":"S-lora: Serving thousands of concurrent lora adapters","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-13T20:44:23.894752Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-06T18:11:16.146190Z"},"links":{"cited_paper":"/paper/2311.03285","citing_paper":"/paper/2507.09019"},"observation_digest":"sha256:948f7658d0d211b745c43adcc42a5f63538368c1d48bdc4db2b402ea986e3fec","observation_id":"c5d486c1-2b71-4d34-973c-666392469670","resolution":{"observed_at":"2026-08-06T18:11:16.146190Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1701.06538","last_updated":"2017-01-23T18:10:00Z","snapshot_observed_at":"2026-08-13T11:35:07.866136Z","submitted_at":"2017-01-23T18:10:00Z","title":"Outrageously Large Neural Networks: The Sparsely-Gated Mixture-of-Experts Layer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1701.06538","snapshot_observed_at":"2026-08-06T18:11:16.045289Z","title":"Outrageously large neural networks: The sparsely-gated mixture- of-experts layer","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-13T20:44:23.894752Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-06T18:11:16.045289Z"},"links":{"cited_paper":"/paper/1701.06538","citing_paper":"/paper/2507.09019"},"observation_digest":"sha256:351489f2399fb922436b977df5672569e5d1d5cad3a4bd82e6b256b3f3944ad9","observation_id":"6861722a-35ce-4626-ab63-0f76e1044013","resolution":{"observed_at":"2026-08-06T18:11:16.045289Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.01318","last_updated":"2023-02-02T18:44:11Z","snapshot_observed_at":"2026-08-13T23:55:57.074762Z","submitted_at":"2023-02-02T18:44:11Z","title":"Accelerating Large Language Model Decoding with Speculative Sampling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.01318","snapshot_observed_at":"2026-08-06T18:11:15.769967Z","title":"Vidur: A Large-Scale Simulation Framework For LLM Inference","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-13T20:44:23.894752Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-06T18:11:15.769967Z"},"links":{"cited_paper":"/paper/2302.01318","citing_paper":"/paper/2507.09019"},"observation_digest":"sha256:ab39362149df14ecffd7b4db5ede4de0592c009642252714cf07cd8fa1ce9d8c","observation_id":"284a4b5b-dfb1-44f0-a2bb-a24dd331d451","resolution":{"observed_at":"2026-08-06T18:11:15.769967Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15792","last_updated":"2024-08-28T13:35:54Z","snapshot_observed_at":"2026-08-12T22:55:53.283832Z","submitted_at":"2024-08-28T13:35:54Z","title":"Efficient LLM Scheduling by Learning to Rank","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15792","snapshot_observed_at":"2026-08-06T18:11:15.858434Z","title":"Efficient llm scheduling by learning to rank","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-13T20:44:23.894752Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-06T18:11:15.858434Z"},"links":{"cited_paper":"/paper/2408.15792","citing_paper":"/paper/2507.09019"},"observation_digest":"sha256:b16199478b9f4360908eb527ae879ea1272c8ce775b9e8e4efc74aacf51383eb","observation_id":"9afccbaa-727d-41b4-8f2c-29152b6b8767","resolution":{"observed_at":"2026-08-06T18:11:15.858434Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-13T20:44:23.894752Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems"},"reference_resolution":{"displayed":12,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":10,"verified_exact":0,"verified_fuzzy":1},"total_outbound_references":12},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 12 of 12 outbound references and 5 inbound Pith citation observations for arXiv:2507.09019."}