{"as_of":"2026-08-09T01:21:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:ab35518bd33d157cc63b54be3f3553d8d9d0df964ec29c578260f2a24406eb4c","coverage":[{"denominator":73,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":73,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T19:04:00.018717Z","state":"measured"},{"denominator":77,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":77,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":4,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":4,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-03T23:39:07.565606Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T06:07:40.898496Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.06608","snapshot_observed_at":"2026-08-03T23:39:07.565606Z","title":"Shi, X., Cai, C., Du, J., and Jia, Z","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2511.04791","last_updated":"2026-05-30T23:27:43Z","snapshot_observed_at":"2026-08-03T23:39:04.093171Z","submitted_at":"2025-11-06T20:18:34Z","title":"DuetServe: Harmonizing Prefill and Decode for LLM Serving via Adaptive GPU Multiplexing","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-03T23:39:07.565606Z"},"links":{"cited_paper":"/paper/2507.06608","citing_paper":"/paper/2511.04791"},"observation_digest":"sha256:5dc481d85987ee8dbd276e3060d392c99c729a0ea4b93c23189cba30565af500","observation_id":"9455ab2d-4938-4969-8633-9e880bf8a1ca","resolution":{"observed_at":"2026-08-03T23:39:07.565606Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"cited_work":{"arxiv_id":"2507.06608","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.06608","snapshot_observed_at":"2026-07-03T06:07:40.898496Z","title":"Nexus: Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","venue":null,"work_id":"582c52e8-2a74-49c4-be74-2d23590374b7","year":2025},"citing_paper":{"arxiv_id":"2606.04415","last_updated":"2026-06-03T03:49:34Z","snapshot_observed_at":"2026-07-06T23:44:33.095008Z","submitted_at":"2026-06-03T03:49:34Z","title":"FlexNPU: Transparent NPU Virtualization for Dynamic LLM Prefill-Decode Co-location","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-28T04:57:08.746551Z"},"links":{"cited_paper":"/paper/2507.06608","citing_paper":"/paper/2606.04415"},"observation_digest":"sha256:35f5a8089853c3de2fb2172c84d9a05c47732b8245ab8c510bf0b86a1d45d425","observation_id":"5d615ec3-c1c4-4a6d-bf8a-55b0ab92bae6","resolution":{"observed_at":"2026-07-02T10:46:52.305907Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"cited_work":{"arxiv_id":"2507.06608","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.06608","snapshot_observed_at":"2026-07-03T06:07:40.898496Z","title":"Nexus: Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","venue":null,"work_id":"582c52e8-2a74-49c4-be74-2d23590374b7","year":2025},"citing_paper":{"arxiv_id":"2607.02043","last_updated":"2026-07-02T11:10:05Z","snapshot_observed_at":"2026-08-08T12:01:49.830379Z","submitted_at":"2026-07-02T11:10:05Z","title":"Towards Load-Aware Prefill Deflection for Disaggregated LLM Serving","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-07-03T06:05:06.467649Z"},"links":{"cited_paper":"/paper/2507.06608","citing_paper":"/paper/2607.02043"},"observation_digest":"sha256:2694cb9fc707971b6e242cd0efeacfc0247b46f955114a0ce2457e7c4c83dbdd","observation_id":"2a000ee2-6084-4102-b695-4e8942c0a79a","resolution":{"observed_at":"2026-07-03T06:07:40.900177Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.06608","snapshot_observed_at":"2026-07-12T09:50:23.266920Z","title":"arXiv preprint arXiv:2507.06608(2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.02574","last_updated":"2026-06-30T16:12:40Z","snapshot_observed_at":"2026-07-12T09:50:22.497230Z","submitted_at":"2026-06-30T16:12:40Z","title":"From Tensor Buffer to Distributed Memory Hierarchy: A Survey of KV Cache Management for LLM Serving","version":1},"reference_index":108,"source":"pdf_text","source_observed_at":"2026-07-12T09:50:23.266920Z"},"links":{"cited_paper":"/paper/2507.06608","citing_paper":"/paper/2607.02574"},"observation_digest":"sha256:3a73833618c8add1df5a00358ca06a82974178d3c65aab669b258f3783ad5846","observation_id":"fb34618a-7af4-44c4-9680-e2f43e418f93","resolution":{"observed_at":"2026-07-12T09:50:23.266920Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2507.06608/citation-record","integrity":"/paper/2507.06608/integrity","json":"/paper/2507.06608/citation-record.json","paper":"/paper/2507.06608"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:11.613769Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":null,"work_id":"0c9017de-6434-45f9-b194-bfbb19f53eb6","year":2024},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:49.235169Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:970907e4c7450d6c42da13e790b07734e527ca8fde9f77ee79d008fa2396df96","observation_id":"338ac59a-682f-4d35-a0bd-f61151c6836f","resolution":{"observed_at":"2026-08-06T19:04:11.736596Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16609","last_updated":"2023-09-28T17:07:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-28T17:07:49Z","title":"Qwen Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16609","snapshot_observed_at":"2026-08-06T19:03:49.510221Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:49.510221Z"},"links":{"cited_paper":"/paper/2309.16609","citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:6a2f0ed74119c2de9a9c6095a0645a8973364060605adc229f3b8f4c39db6744","observation_id":"84795861-3685-46e1-a282-9936e2c205ca","resolution":{"observed_at":"2026-08-06T19:03:49.510221Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:11.366375Z","title":null,"venue":null,"work_id":"3bf7ff95-4961-4c04-929f-a97bd68cf66b","year":2020},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:49.708277Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:9c7c3546fb8ad0cb55e045ae4ceac6edb867e614a0521f2d08b3212dedca78fc","observation_id":"9489a20a-c752-4609-9e00-7dcc9544f230","resolution":{"observed_at":"2026-08-06T19:04:11.486129Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.14165","last_updated":"2020-07-22T19:47:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-05-28T17:29:03Z","title":"Language Models are Few-Shot Learners","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.14165","snapshot_observed_at":"2026-08-06T19:03:49.883433Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:49.883433Z"},"links":{"cited_paper":"/paper/2005.14165","citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:9cea83f49f50727e51b9e5575cd6fd08b43ad9f6f0ea0108b545d4400be2472b","observation_id":"fbdc4538-9768-4d91-89b5-e62fdd8c1426","resolution":{"observed_at":"2026-08-06T19:03:49.883433Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:11.070455Z","title":null,"venue":null,"work_id":"c5c4a4d6-c560-45f2-bd43-2e8b1d2e161d","year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:50.017175Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:f3e2c9d3343ff40603e6ce6e63b3f5a2739055034428ac208468ee5564b575c1","observation_id":"d109e774-c7f5-420f-bc48-17ef72bbf806","resolution":{"observed_at":"2026-08-06T19:04:11.246819Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.13761","last_updated":"2024-10-21T15:59:18Z","snapshot_observed_at":"2026-07-06T19:18:57.862380Z","submitted_at":"2024-09-16T18:46:24Z","title":"Do Large Language Models Need a Content Delivery Network?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.13761","snapshot_observed_at":"2026-08-06T19:03:50.191169Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:50.191169Z"},"links":{"cited_paper":"/paper/2409.13761","citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:0826859824ab45828b6864caa8e6b80aad96e2350e3e241a25283d384c9e279f","observation_id":"e2c367d1-b663-449a-af25-e5c17d432e4f","resolution":{"observed_at":"2026-08-06T19:03:50.191169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:10.754052Z","title":null,"venue":null,"work_id":"d5dd74ee-6357-4f02-9510-0c05b64d07df","year":2022},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:50.320786Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:168ce196ce044c26dc135baccc757f3fb3c226a353874029ba23c1e87f419273","observation_id":"a8228452-758c-4729-a896-14aa2623a3b0","resolution":{"observed_at":"2026-08-06T19:04:10.912794Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:10.570280Z","title":null,"venue":null,"work_id":"3748b211-48c1-4387-b8b6-00becfc710b9","year":null},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:50.468459Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:ada70928d40ea04f70970d0956c491dd7c4068cee07b570a3c7df4fc749f7669","observation_id":"2520886e-153b-45a2-8cf0-78bea8e97e36","resolution":{"observed_at":"2026-08-06T19:04:10.665316Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:10.258889Z","title":null,"venue":null,"work_id":"326ff631-5a45-450d-b8f8-b660be6ea32a","year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:50.810001Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:7540f739f47c40b0ec4f67653ba6a117ff0639c4f0dd2d9575d30bc56d280e09","observation_id":"fba74610-43a8-4058-86a1-b9f02f9a1602","resolution":{"observed_at":"2026-08-06T19:04:10.376948Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:09.977104Z","title":null,"venue":null,"work_id":"8b44af9b-3feb-416f-a610-e06aac3b3c10","year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:50.984123Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:6055d351a9e844d2944b3838aca00b15eb10df4e863d0cd4a4ab315d43cdb2d6","observation_id":"89ad9752-e6c8-4cd4-8cde-f27ca7f5fc7f","resolution":{"observed_at":"2026-08-06T19:04:10.110853Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:51.173248Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:51.173248Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:8d901edb86339752ab09f9dab80cb08fff37f6a1647987b2a5bacf5e6fb0f34f","observation_id":"3faef791-674e-49bb-95a5-a6a03096a9ee","resolution":{"observed_at":"2026-08-06T19:03:51.173248Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:09.685635Z","title":null,"venue":null,"work_id":"038fde69-e671-4ac1-9acd-08455ea1f822","year":2024},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:51.315368Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:c98a9e79c4030cdc550c4929c2e278b5609db08580b16e733bbc4a6e2256639b","observation_id":"df4bd1e9-ae11-4002-be56-a4053e95ceb2","resolution":{"observed_at":"2026-08-06T19:04:09.789661Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-06T19:03:51.523654Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:51.523654Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:b7ed7d27e0bdadcaac7c886a918bfef1f4c51bfcdb832de4eb5544b971624d59","observation_id":"b9085ef6-2eb0-467e-8b1e-315fa0a65278","resolution":{"observed_at":"2026-08-06T19:03:51.523654Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:09.441505Z","title":null,"venue":null,"work_id":"fbce56b2-a68a-4bff-8dad-b618f13cfd63","year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:51.725573Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:d8ed0a703256b609e65cbc1b77ac18fd40f2887dc18c7d077e386969f7112d71","observation_id":"4038bd45-7ff2-4fbe-afb2-47d4fdaca41f","resolution":{"observed_at":"2026-08-06T19:04:09.532396Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:09.118881Z","title":"2023.Amazon Found Every 100ms of Latency Cost them 1% in Sales.https://www.gigaspaces.com/blog/amazon-found-every- 100ms-of-latency-cost-them-1-in-salesAccessed: 2025-05-28","venue":null,"work_id":"77c5901e-1a48-45ff-98e6-54a2f6abe1ef","year":2023},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:51.876357Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:1192a51e3e6e81bfa7f8378112bf7defe461bacc240ddca993fc1dde59a73fc8","observation_id":"f61bf53d-a664-456e-b84d-0a79e29da4a7","resolution":{"observed_at":"2026-08-06T19:04:09.286762Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:52.061942Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:52.061942Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:7f62718880a75780b01408ce05f5918363c69885c4a1e8bdb81cb49d7ce18552","observation_id":"9a79cb2d-b83d-473c-8ad0-b0c9181ed04f","resolution":{"observed_at":"2026-08-06T19:03:52.061942Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:08.834230Z","title":null,"venue":null,"work_id":"b3b1a1dd-a5d4-40f2-a099-a115a92f0baf","year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:52.275100Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:6dd83d69ae955a5a3273ae2b05c88248bb417a0a338546c457cae55b16ebc81a","observation_id":"44f04314-4f26-43e9-8ca0-e57f277068cb","resolution":{"observed_at":"2026-08-06T19:04:08.947581Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-06T19:03:52.402578Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:52.402578Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:f935fe8e12c9291bf5bc24443b52549c8bc73fffbd0cbbd1e38f232c73828867","observation_id":"fab68d91-9ac4-4ab2-b723-23d813a0928f","resolution":{"observed_at":"2026-08-06T19:03:52.402578Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.14196","last_updated":"2024-01-26T09:23:11Z","snapshot_observed_at":"2026-08-06T22:40:28.707813Z","submitted_at":"2024-01-25T14:17:53Z","title":"DeepSeek-Coder: When the Large Language Model Meets Programming -- The Rise of Code Intelligence","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.14196","snapshot_observed_at":"2026-08-06T19:03:52.557111Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:52.557111Z"},"links":{"cited_paper":"/paper/2401.14196","citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:5a9c72d49a43f0d5a8e4c5f90996d7f93f42c8bb3be7424195d75a8a8b86b276","observation_id":"82245741-c021-40c4-8f19-0286ccfbef88","resolution":{"observed_at":"2026-08-06T19:03:52.557111Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:08.604237Z","title":null,"venue":null,"work_id":"74569801-55be-4835-ac23-1034d141975e","year":2022},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:52.707509Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:77dc1c6257798fe26e278a6715096644f27d6c61ad8340c76faaa67b40f2e904","observation_id":"476b1acb-ffb8-4166-b714-e13be606dffe","resolution":{"observed_at":"2026-08-06T19:04:08.718233Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.19867","last_updated":"2025-04-28T15:00:03Z","snapshot_observed_at":"2026-08-07T15:58:35.107523Z","submitted_at":"2025-04-28T15:00:03Z","title":"semi-PD: Towards Efficient LLM Serving via Phase-Wise Disaggregated Computation and Unified Storage","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.19867","snapshot_observed_at":"2026-08-06T19:03:52.902518Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:52.902518Z"},"links":{"cited_paper":"/paper/2504.19867","citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:fbf216f3acdb2d44a49e106b8e79fadee4103354f12b64d384ed62016da69175","observation_id":"fd7d91d0-44e7-4401-95ab-fb53c88276eb","resolution":{"observed_at":"2026-08-06T19:03:52.902518Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.11181","last_updated":"2024-01-20T09:43:36Z","snapshot_observed_at":"2026-08-05T23:35:16.350557Z","submitted_at":"2024-01-20T09:43:36Z","title":"Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.11181","snapshot_observed_at":"2026-08-06T19:03:53.051327Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:53.051327Z"},"links":{"cited_paper":"/paper/2401.11181","citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:7428f299af2edfdc1b8d4828f0c2e8d82ecf4b92efd085791382b439195372a2","observation_id":"5f148349-ce14-4b64-8106-9106552df42e","resolution":{"observed_at":"2026-08-06T19:03:53.051327Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:53.207078Z","title":"Kamath, Ramya Prabhu, Jayashree Mohan, Simon Peter, Ramachandran Ramjee, and Ashish Panwar","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:53.207078Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:93f72e3bfe33a7303b0b66988bf2671c9c06426f02e85cdccfbda99169e61e23","observation_id":"8bba5241-8d10-445c-bf9e-e76c27467558","resolution":{"observed_at":"2026-08-06T19:03:53.207078Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:53.363867Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:53.363867Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:acee990fd393b609157c3577f2fa1aad6cc2aeb1f9ad68a919d860b004400d50","observation_id":"83d732ee-6279-481a-ac41-5c63f96ad675","resolution":{"observed_at":"2026-08-06T19:03:53.363867Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:08.256321Z","title":null,"venue":null,"work_id":"7bfb37fb-467d-4777-ae08-196443f589bc","year":null},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:53.700359Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:2d922549b3e0537d30502d2ff3e296bf769f3c11853ddbc9296fe7460879d6c2","observation_id":"e485f8f8-5398-4ad0-893f-69dc854f9fcd","resolution":{"observed_at":"2026-08-06T19:04:08.425322Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-06T19:03:54.059640Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:54.059640Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:77008558a9e6b57b907eacc4f223023ac38e5556fb12e5ac71f8bef459d8f896","observation_id":"ab36e2cd-84c7-461b-a318-962b2eadc62e","resolution":{"observed_at":"2026-08-06T19:03:54.059640Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:54.252342Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:54.252342Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:ec08c7b5fab7511366badd68311c14191a19b2d0bc355acad1d83a04ef4d2104","observation_id":"14fa65db-8dae-4679-ae49-2d9fb31f77c1","resolution":{"observed_at":"2026-08-06T19:03:54.252342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:53.849594Z","title":"arXiv:2504.19516 [cs.DC]https:// arxiv.org/abs/2504.19516","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:53.849594Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:f202805276b26edab6ba3b79b56137cb5f4d67245ddeff1b09162baab73cb8cb","observation_id":"6b9fd67d-b60b-4a73-8615-a3cede4c2c89","resolution":{"observed_at":"2026-08-06T19:03:53.849594Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:54.532273Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:54.532273Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:4e7a1eae9a6b7cc3a2932cdf29448f0c769b5cfefd96ec70b9ef92e164d50a33","observation_id":"9e550a0e-44b1-4e96-a1dc-7180d7205cc5","resolution":{"observed_at":"2026-08-06T19:03:54.532273Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:07.708426Z","title":null,"venue":null,"work_id":"6ea3bb7c-e2bd-4984-b175-fa7675562a35","year":2024},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:54.693991Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:b55f79e1dc173ff1844c9908c49f53d8b0c89073abaa1768a9a2bb3b6e5d0d58","observation_id":"2cc7525e-a3b5-4822-a188-fd9469b3a89a","resolution":{"observed_at":"2026-08-06T19:04:07.824320Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:07.973288Z","title":null,"venue":null,"work_id":"eaa7408b-6412-4027-b53d-e383e02335d3","year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:54.394084Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:3d308fedeff301e3ca38bee9b39923fd66e16b6580c173976d575adcdc504ab4","observation_id":"4dc625e5-7919-431a-9372-cb5fb079e2c6","resolution":{"observed_at":"2026-08-06T19:04:08.122381Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:07.364105Z","title":null,"venue":null,"work_id":"51a299b5-912f-4f18-b2f6-e43dc0c9598d","year":2024},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:55.020878Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:5f3728779474e5e5455d2de6d56fcc181105f15d63ff571bad2cf4153ec82733","observation_id":"0341b78a-2809-4d17-9d73-6c694c8e1b5f","resolution":{"observed_at":"2026-08-06T19:04:07.531451Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:07.177324Z","title":null,"venue":null,"work_id":"a6ede6db-8ec0-4372-a7e0-9eae04c8250e","year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:55.166015Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:7cbeccf5c3be7cd8f9aed6a8501c3bcc633202fc977fd7a06e6bb90b73bcf1a7","observation_id":"357a2d89-9797-4c1d-a41a-59ca630c5b28","resolution":{"observed_at":"2026-08-06T19:04:07.271974Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:54.844157Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:54.844157Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:783be76534fe83e9ca10f6e59d823d824b4fc29d4dc8d231d397e4fb842763aa","observation_id":"379a4380-abc8-484c-bef8-3df4665291fa","resolution":{"observed_at":"2026-08-06T19:03:54.844157Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:06.687381Z","title":null,"venue":null,"work_id":"00e18600-8f39-4d2a-a6ca-52d43183ff3d","year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:55.408720Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:bf7b85b046906cd3cf8b968bdcce4129c1055103afd9f7325372171a6c968ad9","observation_id":"e432d991-c8ce-4d9d-bfb3-1d0cc699e9e2","resolution":{"observed_at":"2026-08-06T19:04:06.825929Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:06.371691Z","title":null,"venue":null,"work_id":"1d2403eb-a82d-476a-8452-f25f8f05b01b","year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:55.536989Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:961d124e2ffd9e50bd6345e333cfdc08f74e7c5b5caca32e5457cb8374f76401","observation_id":"bf72e16b-9bfa-41bf-8963-5629e50af8df","resolution":{"observed_at":"2026-08-06T19:04:06.546283Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:06.991059Z","title":null,"venue":null,"work_id":"41fa36c3-9dec-453d-9c2e-13d1190e08ac","year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:55.331604Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:cd27420c2d87818a09d4dd5603aa041f26799759c2d82d574d970cc9d42a4399","observation_id":"0e9a40cf-b2d4-41d8-acbf-98f36c9b9e82","resolution":{"observed_at":"2026-08-06T19:04:07.073341Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:55.779394Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:55.779394Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:ef02dff20ef1b2585bdc4e0e9f17ec755586e7f28b6fef779993d3833eb955b3","observation_id":"43083630-f123-4726-988b-8ad9c6431e63","resolution":{"observed_at":"2026-08-06T19:03:55.779394Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-06T19:03:55.916753Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:55.916753Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:3b62aeac982517b745aa2a5f1545efd78704a3a6b0f2a8d43921e32479a3eaad","observation_id":"acf75792-50d4-44f1-94cc-725cb077c48f","resolution":{"observed_at":"2026-08-06T19:03:55.916753Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:06.088998Z","title":null,"venue":null,"work_id":"c8df820c-eece-4c24-8952-542b05d1f9d5","year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:55.666466Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:5a767a5bd3bc42226dfe8f78738dd16e9911445dba175318606a5b04d332997c","observation_id":"bcdf12a6-12e0-4117-8a08-a693da0fa9d8","resolution":{"observed_at":"2026-08-06T19:04:06.242356Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:56.139036Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:56.139036Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:9008b429a4e18ff663db0dd40fb3aad666acfda54bc711f1ad5ba95a37a7df53","observation_id":"0b6d9e27-2d37-4421-85e8-42c30561b000","resolution":{"observed_at":"2026-08-06T19:03:56.139036Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:05.476436Z","title":null,"venue":null,"work_id":"2d2e8f50-266c-48e5-968c-11b4cc5ed10b","year":2024},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:56.295673Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:a200d794da137d6982d21d21e763eda5f48400b0c0b84afdeee919f1164ee0e9","observation_id":"bc1e240a-daf9-403d-bb9e-77cf95a5cc7d","resolution":{"observed_at":"2026-08-06T19:04:05.604874Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:05.767428Z","title":null,"venue":null,"work_id":"96a2e71e-e7b3-490d-8ddb-1d8bef1fcfe4","year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:56.029991Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:ebbbbc92d5d47bc08316214f22af6c04ef1e686b18ed8dd6f4e7f0137099fc56","observation_id":"5e67b348-d6e0-4c1f-96e4-e5e0ab6e8162","resolution":{"observed_at":"2026-08-06T19:04:05.939629Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:56.686325Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:56.686325Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:c4fd7c3d47c185115380854957418fb0ac30a4498543c8d100faaca2ff68bc85","observation_id":"6a4b2725-bb70-41fd-b78d-04f6e1ab51cc","resolution":{"observed_at":"2026-08-06T19:03:56.686325Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:04.964573Z","title":null,"venue":null,"work_id":"9522029d-e2b9-4c0b-b594-f43b0dfd819a","year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:56.851205Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:324ad0724b6fd961434fd489b99dc50438bd6358ebac6caf11e180e4cb9e5d89","observation_id":"60410545-a773-4b16-95b1-32fc83e05e64","resolution":{"observed_at":"2026-08-06T19:04:05.069178Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:04.716961Z","title":"Gonzalez, and Ion Stoica","venue":null,"work_id":"d2d6c903-344d-4196-9785-45373a7898a6","year":2024},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:56.966878Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:e1bc71777aa63efed626a637e26514f9e8d2c1d46987a8ceeb7fd8476a934690","observation_id":"1ef2e1ec-ed26-4738-9503-29f1e1160c44","resolution":{"observed_at":"2026-08-06T19:04:04.842322Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:05.213077Z","title":null,"venue":null,"work_id":"d2bc906e-c068-425e-b7a6-407a660a6e38","year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:56.574785Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:650c918dcc1987df1b517f9c27e49bcdd357f4f8e0a04b1ee473bd96aad83ea8","observation_id":"6373584b-1483-4fa2-96f7-46214218cb5b","resolution":{"observed_at":"2026-08-06T19:04:05.336503Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:57.233230Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:57.233230Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:ef198ac469869e4e1a6b99e098a0bb7da33e8b30127e2ca9ec4e8415558657d0","observation_id":"20fefb6d-eb13-4f2f-9d81-498c9956a2b3","resolution":{"observed_at":"2026-08-06T19:03:57.233230Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:04.473079Z","title":null,"venue":null,"work_id":"d219b89f-4e27-4aaf-ab7e-3893a30bfd58","year":2024},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:57.353411Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:30aae3cef1d4b6f6df7080b98dcf681ee55455520a9b61a7c18131d5e7cb192a","observation_id":"98935549-5dcc-4674-9567-4c4fc2617f63","resolution":{"observed_at":"2026-08-06T19:04:04.598001Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:04.249608Z","title":null,"venue":null,"work_id":"94227eca-278a-4bdd-b9ec-073fbcc7862b","year":null},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:57.491862Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:e1b62888283a7fecac05da96b393452c6bcccf0ede9b10428f1a27dd5e021833","observation_id":"9328e727-9bcd-4b66-a854-94cd0e4e7ce6","resolution":{"observed_at":"2026-08-06T19:04:04.346503Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00023","last_updated":"2024-10-03T17:50:33Z","snapshot_observed_at":"2026-08-03T09:39:42.593902Z","submitted_at":"2024-05-08T06:30:58Z","title":"Preble: Efficient Distributed Prompt Scheduling for LLM Serving","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.00023","snapshot_observed_at":"2026-08-06T19:03:57.101107Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:57.101107Z"},"links":{"cited_paper":"/paper/2407.00023","citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:63a9182fa533d42aa93e643d71f0091e8738bc7676eded92c826fb54b237e51b","observation_id":"914c9eb8-8f75-475d-8c57-35b59e533f5e","resolution":{"observed_at":"2026-08-06T19:03:57.101107Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:03.689959Z","title":null,"venue":null,"work_id":"d53cddbc-bc5d-4bbc-bede-e03057182f0f","year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:57.887291Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:54c6fd4f6597b9439d705aa4313fab5702ef876ff5bf83b15faf46ccfb58b8f2","observation_id":"09a61544-afd7-4c8f-9749-39548c2ed5ef","resolution":{"observed_at":"2026-08-06T19:04:03.818694Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:03.391538Z","title":"Gomez, Lukasz Kaiser, and Illia Polosukhin","venue":null,"work_id":"40b0f2bb-73d4-43e8-865b-c1bdebea4b4f","year":2017},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:58.008600Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:4b4b7e12a6c977a4630dffa2f02d3f32089799de645f2222ed97c9a3973ece0a","observation_id":"f645e1bd-897d-4f45-bcac-bab857170173","resolution":{"observed_at":"2026-08-06T19:04:03.543498Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:03.078447Z","title":null,"venue":null,"work_id":"39dd264b-2376-4c2a-8408-4d858f63f705","year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:58.152919Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:749f4cc54d751d054c150a1662faae82dacc554edc5cd510b1eb6c88fe5209a5","observation_id":"81fdbb97-0a95-4613-8daa-eefad67f0317","resolution":{"observed_at":"2026-08-06T19:04:03.219656Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12599","last_updated":"2025-06-03T02:14:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T02:48:14Z","title":"Kimi k1.5: Scaling Reinforcement Learning with LLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12599","snapshot_observed_at":"2026-08-06T19:03:57.605849Z","title":"5: Scaling reinforcement learning with llms.arXiv preprint arXiv:2501.12599(2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:57.605849Z"},"links":{"cited_paper":"/paper/2501.12599","citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:e21e4f36cd3522b3ccccef468cf60738ff7bc45163022979b8e990c92860e379","observation_id":"6784f347-aede-4aac-903a-49029752aa6f","resolution":{"observed_at":"2026-08-06T19:03:57.605849Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:03.957977Z","title":null,"venue":null,"work_id":"66cc4d2d-0086-4bf6-8e01-64ff658ccd94","year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:57.762636Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:ac3d8d62c4fa2e1c43dbc2efca8fb56642a3285b3d0bf91b88e4ef2ea4bef3d8","observation_id":"640876cc-0e1a-49d7-8f3d-8721163250a1","resolution":{"observed_at":"2026-08-06T19:04:04.126560Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-08-06T19:03:58.507968Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:58.507968Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:d9c58651d944a79e1db5937599f3ab402c9064943f2d50c823241a726e583a06","observation_id":"eb3a8150-5dbd-426b-8292-69ee78cee1f5","resolution":{"observed_at":"2026-08-06T19:03:58.507968Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-06T19:03:58.622896Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:58.622896Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:d22edc2ea3ac8228f14d35311f01c954124e42782e7a96f1ff038d84cb41d7ab","observation_id":"db0b4271-7b89-402b-9261-07051ce187e1","resolution":{"observed_at":"2026-08-06T19:03:58.622896Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.16444","last_updated":"2025-04-03T22:49:22Z","snapshot_observed_at":"2026-08-07T00:11:38.883613Z","submitted_at":"2024-05-26T06:00:17Z","title":"CacheBlend: Fast Large Language Model Serving for RAG with Cached Knowledge Fusion","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.16444","snapshot_observed_at":"2026-08-06T19:03:58.768555Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:58.768555Z"},"links":{"cited_paper":"/paper/2405.16444","citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:9660d52351838df61905efa5c3ac7b18f73908a8df92b2ac6b40474ac75db66f","observation_id":"7dc183ed-5565-4bb9-beb1-5f588b8fee2d","resolution":{"observed_at":"2026-08-06T19:03:58.768555Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05920","last_updated":"2024-09-25T05:57:51Z","snapshot_observed_at":"2026-07-06T15:25:24.491136Z","submitted_at":"2023-05-10T06:17:50Z","title":"Fast Distributed Inference Serving for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.05920","snapshot_observed_at":"2026-08-06T19:03:58.274083Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:58.274083Z"},"links":{"cited_paper":"/paper/2305.05920","citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:dffe81ec2a470250746a959af08ccc0480edf3b8af70193ee1518be8ac81ee2c","observation_id":"222941a3-4027-496c-b9d8-bddb8dc2bde2","resolution":{"observed_at":"2026-08-06T19:03:58.274083Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:02.772747Z","title":null,"venue":null,"work_id":"34278f39-db61-452d-8a87-255823491ec3","year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:58.388520Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:7ec1bbc1ee4141b78d42eeab1eea09cc0c22b8d6976449d1affc143771e446b9","observation_id":"d49a8f5f-e8a9-48fd-a17b-81ddcefca7f6","resolution":{"observed_at":"2026-08-06T19:04:02.910222Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:59.157228Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:59.157228Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:be51cd55581da5c3f5008f53f1496cd516011842d16dd0e041b1fc05349fdc96","observation_id":"c13ea7ff-6824-42c7-b1b2-8e0ab499dd84","resolution":{"observed_at":"2026-08-06T19:03:59.157228Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:02.204148Z","title":null,"venue":null,"work_id":"3265febb-6873-4873-b3c0-b839ae1c3bcb","year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:59.286492Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:5625fc94084537e6faa8c19dd2787eb490ea2f584ecda2a631f2591f47c4c31b","observation_id":"94c17921-321e-4b32-868e-d42da56c5145","resolution":{"observed_at":"2026-08-06T19:04:02.348232Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:59.420592Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:59.420592Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:40a8445704e42cc818e0d4eabfc0e2abca25c45076bce02fbfe3483f06060bae","observation_id":"dc8fc7bb-2538-4f7f-b1ec-6ccd308ef320","resolution":{"observed_at":"2026-08-06T19:03:59.420592Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01005","last_updated":"2025-04-21T20:10:11Z","snapshot_observed_at":"2026-07-06T20:15:36.280948Z","submitted_at":"2025-01-02T02:02:20Z","title":"FlashInfer: Efficient and Customizable Attention Engine for LLM Inference Serving","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.01005","snapshot_observed_at":"2026-08-06T19:03:58.879397Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:58.879397Z"},"links":{"cited_paper":"/paper/2501.01005","citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:6fd8516bcf0b040440a1ff4bfcb72495480c6dd888b960ed432039cb18dc1523","observation_id":"2f8bc3fc-5475-47bf-b506-3556af9a1f0e","resolution":{"observed_at":"2026-08-06T19:03:58.879397Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:02.501589Z","title":null,"venue":null,"work_id":"6431c1f7-2fdf-4eea-9a4b-444defd8c748","year":2022},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:59.000796Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:57c3dc023f0ffb802e9eaa0075fd6226bd7a5e9d4061a37894cac18af323f331","observation_id":"690619c1-3fe7-41e6-91dd-b82600bce5d1","resolution":{"observed_at":"2026-08-06T19:04:02.625783Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:01.759768Z","title":null,"venue":null,"work_id":"29442c6a-d1a2-45be-b4e0-0a2fa7037877","year":null},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:59.868498Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:6d8f94599866174d58dcb4a8b8aa5e59a2bee7bbdb6a04bea180e3a04df2dc95","observation_id":"0d7c2f59-f5eb-48c7-b152-45a7191492f0","resolution":{"observed_at":"2026-08-06T19:04:01.891298Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.07104","last_updated":"2024-06-06T00:10:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-12T09:34:27Z","title":"SGLang: Efficient Execution of Structured Language Model Programs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.07104","snapshot_observed_at":"2026-08-06T19:03:59.548009Z","title":"Gonzalez, Clark W","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:59.548009Z"},"links":{"cited_paper":"/paper/2312.07104","citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:8f3c0beb2a3d8bbbd38826d9b1eec2726e89aaacd24031b36c1ea26d05a2c0b3","observation_id":"d4f400db-7cdb-4c9a-a6fb-2177cc26c0ca","resolution":{"observed_at":"2026-08-06T19:03:59.548009Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:02.014259Z","title":null,"venue":null,"work_id":"96eaffc7-ebdd-4120-b0ef-ee640b2e0406","year":2024},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:59.705803Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:990a783d8a8b500034ec0bc15e920fd96c5d2517b1a833f42b03c36c63a1a649","observation_id":"19dca056-fc0a-465f-9aeb-93281f4a7b8c","resolution":{"observed_at":"2026-08-06T19:04:02.084309Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12757","last_updated":"2025-05-25T14:08:01Z","snapshot_observed_at":"2026-08-03T08:33:17.421277Z","submitted_at":"2024-08-22T23:00:40Z","title":"NanoFlow: Towards Optimal Large Language Model Serving Throughput","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12757","snapshot_observed_at":"2026-08-06T19:04:00.018717Z","title":"doi:10.48550/ARXIV.2408","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-06T19:04:00.018717Z"},"links":{"cited_paper":"/paper/2408.12757","citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:d629784fe56870f71a7c47beb63ecc6cee66a34f83b7787e7f4dc8caae5c420f","observation_id":"a275db72-d7ee-4520-bd53-7035783ed824","resolution":{"observed_at":"2026-08-06T19:04:00.018717Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:53.569435Z","title":"InProceedings of the 29th Symposium on Operating Systems Principles, SOSP 2023, Koblenz, Germany, October 23-26, 2023, Jason Flinn, Margo I","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:53.569435Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:356abd615620afb3fe8db870253b24814bc785eb53b1637cf21c5210c30db8a2","observation_id":"0b28cfbb-988a-4c19-b601-ad0b681061c7","resolution":{"observed_at":"2026-08-06T19:03:53.569435Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:56.457597Z","title":"doi:10.1145/3698038.3698523","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:56.457597Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:ae18d2d24ca1d788db4ce68b47b66554a6777b8302faf7703340a34db13f27a1","observation_id":"a3d35633-61f7-4514-9373-0f50becc89e5","resolution":{"observed_at":"2026-08-06T19:03:56.457597Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.48550/arxiv.2504.14489","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:04:00.391512Z","title":"CoRRabs/2504.14489 (2025)","venue":null,"work_id":"ec85271c-165f-41a0-9b28-75693965f119","year":2025},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:50.620327Z"},"links":{"citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:45304074c45289cfdc51821e6b1db3c0fa2f120da6f7e5aec0041d6fcaa296c9","observation_id":"1bafd17a-0b90-411d-97ef-50b1697fb5c6","resolution":{"observed_at":"2026-08-06T19:04:00.469184Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","latest_version":5,"primary_category":"cs.DC","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving"},"reference_resolution":{"displayed":73,"state_counts":{"malformed_identifier":2,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":66,"verified_exact":1,"verified_fuzzy":4},"total_outbound_references":73},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 73 of 73 outbound references and 4 inbound Pith citation observations for arXiv:2507.06608."}