{"as_of":"2026-08-19T03:37:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:73b15279e4cf4c2c27be4c78bd61b3f55905c3d2b446057879c7e17036a115b3","coverage":[{"denominator":36,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":36,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T10:18:24.943745Z","state":"measured"},{"denominator":37,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":37,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-14T20:35:48.828322Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.05871","snapshot_observed_at":"2026-07-14T20:35:48.828322Z","title":"Bestserve: Serving strategies with optimal good- put in collocation and disaggregation architectures.CoRR abs/2506.05871(2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.15202","last_updated":"2026-06-05T04:55:52Z","snapshot_observed_at":"2026-08-15T22:55:45.942058Z","submitted_at":"2026-03-16T12:43:32Z","title":"Simple is Better: Multiplication May Be All You Need for LLM Request Scheduling","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-07-14T20:35:48.828322Z"},"links":{"cited_paper":"/paper/2506.05871","citing_paper":"/paper/2603.15202"},"observation_digest":"sha256:0a73b063a4f5b9e83312fbf383a04fd68cdc1be51c699a52b5637ada1deace6a","observation_id":"23e4ae32-5d5a-4207-aa2e-5810f04e42cb","resolution":{"observed_at":"2026-07-14T20:35:48.828322Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2506.05871/citation-record","integrity":"/paper/2506.05871/integrity","json":"/paper/2506.05871/citation-record.json","paper":"/paper/2506.05871"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:21.975462Z","title":"Gqa: Training generalized multi-query transformer models from multi-head checkpoints, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:21.975462Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:9952ff08fcb2faba1a34660dd5a04b315a8effc30c858be40a901434aafc93b0","observation_id":"9a84f048-fc01-47c7-bdd5-71fb2a72c78c","resolution":{"observed_at":"2026-08-07T10:18:21.975462Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:31.975345Z","title":"How continuous batching enables 23x throughput in LLM inference while reducing p50 latency","venue":null,"work_id":"e921f029-d308-45b9-88f7-3097058eb653","year":2023},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:22.047032Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:9f34c8beda4253810da60bc5e81f6eb0f4e2eac9f4825949bfa7570fcb917f75","observation_id":"b76c699d-04ab-4a89-9df4-bb88f9d7a139","resolution":{"observed_at":"2026-08-07T10:18:32.117041Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:31.763102Z","title":null,"venue":null,"work_id":"f7f55a79-8551-4dd7-a253-b1ff037f4a97","year":null},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:22.111097Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:5e409a0afb842b8b94c8d19202042d7ace5d2d2ee55974ff6466f4c59e25d016","observation_id":"f116b98f-bafb-43fe-a337-bb60bf1d77e5","resolution":{"observed_at":"2026-08-07T10:18:31.863421Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:22.166883Z","title":null,"venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:22.166883Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:f608fc8fe5e68625a990b23aaf4e7f591025466192ed4d4ac4462a30271690d3","observation_id":"d3569a65-42ef-4074-a853-8884f11aae61","resolution":{"observed_at":"2026-08-07T10:18:22.166883Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:31.463964Z","title":"Throughput is not all you need: Maximizing goodput in llm serving using prefill-decode disaggregation.https://hao-ai- lab.github.io/blogs/distserve/, 2024","venue":null,"work_id":"c056b4f8-aec4-4a3d-bb98-f98df993c232","year":2024},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:22.238130Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:93a09cee5bb680b29db82192ec40b996ec8fb3c1cc24f15a3bb7ae8e942f8be6","observation_id":"4ffec0d6-76ce-4efd-934c-30698fb7ab4c","resolution":{"observed_at":"2026-08-07T10:18:31.602003Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:31.207253Z","title":null,"venue":null,"work_id":"52506093-6cee-485d-842c-e2103a3d0028","year":null},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:22.310379Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:ada60a8bb2215e6ddd97e66d6745f8f4d9972a1fdd4ba3da3491516626fe6e3b","observation_id":"3a7fe4fb-4914-4440-b118-d8163dfad57a","resolution":{"observed_at":"2026-08-07T10:18:31.356513Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:30.943286Z","title":null,"venue":null,"work_id":"79073874-02f6-498f-99e5-9d36bdc27251","year":null},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:22.376684Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:7198664042a5bccbab4a091ff2d00137146114a036b82dc1e63119a56b89d8db","observation_id":"f8c79ad7-811b-472b-ab2e-bb55a4b7f564","resolution":{"observed_at":"2026-08-07T10:18:31.052695Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:30.722727Z","title":null,"venue":null,"work_id":"6e140df6-8ce0-429a-82df-c1d948da4913","year":2022},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:22.458150Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:28819149ea576ac3c8b18675c6fbcebd2bf544a4af9bf05fcb11c593e4b3e0dc","observation_id":"32076b1c-5f97-41ab-ab8d-614d6ea9393c","resolution":{"observed_at":"2026-08-07T10:18:30.817164Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:22.540447Z","title":"Sigmoid-weighted linear units for neural network function approximation in reinforcement learning, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:22.540447Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:0c4c53665f5f4c2854fae8fc31835bfdf6cd97b09e844e16cfcdcbe3cf816732","observation_id":"24b96557-fb56-44a3-95ff-7b54f2e5471a","resolution":{"observed_at":"2026-08-07T10:18:22.540447Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:30.467478Z","title":"Text generation inference","venue":null,"work_id":"9745a484-91cc-4d51-b4ba-ff0bb2ee7b78","year":null},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:22.605821Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:96c45da8ca009022ff656e9b2e6e0e3231cdd0f186ce6cf695a99e3eeeafd3cd","observation_id":"c16b42c3-e6aa-402a-918d-71d2ccaafcb1","resolution":{"observed_at":"2026-08-07T10:18:30.599741Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:30.157260Z","title":"Low latency rnn inference with cellular batching","venue":null,"work_id":"6d048390-5939-40a9-8fc2-48104d4ca736","year":2018},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:22.666922Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:6cebe6642f6b84717c21db786315e9ab1db9dce23b3bbfbcef7054025394c0c2","observation_id":"6e795329-54e9-4350-9bd8-073f3c64a412","resolution":{"observed_at":"2026-08-07T10:18:30.273233Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:29.924463Z","title":"Getting started with CUDA graphs","venue":null,"work_id":"5d585c26-1361-4674-92c5-0848d8620754","year":2019},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:22.721275Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:3a30345e5261c10482d2347507f6fdb83ea2d8eb0ddb9d94c6917a047e6fe7d4","observation_id":"d676dfcc-085d-457a-8234-0659cbf90478","resolution":{"observed_at":"2026-08-07T10:18:30.035124Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:29.606295Z","title":"Shortle, James M","venue":null,"work_id":"173ffa6a-08bb-4833-a8fd-ca9b9e5e6535","year":2008},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:22.782207Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:e2492e86bf6fa072c7d64b8071b06f25ec3ae68f7a4d33904c7c0b0fb056298e","observation_id":"86bb2b9d-0fd9-496c-b9fb-290bcd0678d7","resolution":{"observed_at":"2026-08-07T10:18:29.734840Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:29.359822Z","title":"Pipedream: Fast and efficient pipeline parallel dnn training, 2018","venue":null,"work_id":"2c89ca97-367b-4712-8488-44da797743a4","year":2018},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:22.842703Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:aba1e2959add2168ca1a6cedb8fa635017eab6618b1f9ddbd451b4588debf1f6","observation_id":"0aee096f-153b-446d-b9b7-33c76b19bbb1","resolution":{"observed_at":"2026-08-07T10:18:29.478952Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:22.918596Z","title":"Inference without interference: Disaggregate llm inference for mixed downstream workloads, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:22.918596Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:9b4765cb823975c39fe66f381513d0e3a9f37ba51a103b8618f02b468732759f","observation_id":"f32f8b3b-45e4-4fcc-8ad3-536d46c7d48a","resolution":{"observed_at":"2026-08-07T10:18:22.918596Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:29.117816Z","title":"Le, Yonghui Wu, and Zhifeng Chen.GPipe: efficient training of giant neural networks using pipeline parallelism","venue":null,"work_id":"c86c6722-1c14-4c41-bdb0-72d42a5a71b5","year":2019},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:23.004203Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:853a83e91040a532d551d65a4112c7097d9578b941e42d7ab643a5c0a16dc359","observation_id":"fc0b7dfd-3473-4204-aaa8-8d8ad4ed900b","resolution":{"observed_at":"2026-08-07T10:18:29.186315Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:23.064579Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:23.064579Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:24f9517c4c111de537a82125127bcafa3afbeaa04a8142e62384e7e4fa10abb2","observation_id":"6a9ad0c6-582b-4c8c-bf15-b812ea35e28e","resolution":{"observed_at":"2026-08-07T10:18:23.064579Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:29.064604Z","title":"Efficient memory management for large language model serving with PagedAttention","venue":null,"work_id":"cf6e0603-334f-413a-851d-ed43c20fed44","year":2023},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:23.122706Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:7eab905cc385b9ced92386b22a1c0c11e42b92ddc2d42ff98e6a69e576d09878","observation_id":"145f5443-f828-4631-87b0-8ea6a3de16bb","resolution":{"observed_at":"2026-08-07T10:18:29.105821Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:28.879122Z","title":"Transformers KV caching explained","venue":null,"work_id":"e5d5ae6b-5ac9-4d8a-abcf-b0d37f56b6c3","year":2023},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:23.182303Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:b49a797f8e6a04649337e2efd7967765bc1e61b06da2592bdb867d8bffbc92d0","observation_id":"0d23d8a8-73ad-40d6-91da-9b8aedc2cdec","resolution":{"observed_at":"2026-08-07T10:18:29.006858Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:28.660771Z","title":"Sequence parallelism: Long sequence training from system perspective, 2022","venue":null,"work_id":"a894bff3-67e6-4c76-b8ab-95f7dbc7e4ee","year":2022},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:23.303096Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:4bcc50cc52a296d87c1ab6ca57a1ca54ab00ea228951af37b90792ce36755200","observation_id":"45caf8d9-9714-4566-b9fa-0db1dcdb3cea","resolution":{"observed_at":"2026-08-07T10:18:28.761856Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:28.426744Z","title":"Llama 3.2: Revolutionizing edge AI and vision with open, customizable models","venue":null,"work_id":"e4329bd4-ec3a-4e1f-b51a-c09eb8805fe0","year":2024},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:23.353733Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:f1ebd3a21e0411627d4db5e0a19ac26dbcbda8b47c579dec829e8e7796169359","observation_id":"64e64901-3b20-4764-b6a6-651291e110b3","resolution":{"observed_at":"2026-08-07T10:18:28.549453Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:28.230288Z","title":"NVIDIA TensorRT-LLM.https: //docs.nvidia.com/tensorrt-llm/index.html","venue":null,"work_id":"20b642e6-11b6-4d92-a1f6-b37e0eb574c4","year":null},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:23.468448Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:31c40a9ae1febc722cc3b7654a39f91bc70c7a66110b6f982f858ff53c696907","observation_id":"e8618f5a-0dd0-46b8-a1b4-6c6368e7b5cd","resolution":{"observed_at":"2026-08-07T10:18:28.335252Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:27.955936Z","title":"OpenAI o3-mini","venue":null,"work_id":"d9ffedad-bf4d-4e80-8203-476cda711da3","year":2025},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:23.581221Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:ef571cd22b0d8dd6f9e91e56417abd21ac8308b04661c8e12db3e0851fb01ae8","observation_id":"bc6c61ce-9f85-444a-8f49-441f6dd4c07c","resolution":{"observed_at":"2026-08-07T10:18:28.084574Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:27.702753Z","title":"Splitwise: Efficient generative llm inference using phase splitting, 2024","venue":null,"work_id":"ad471856-02b9-4a5a-9825-ea93aeb269b5","year":2024},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:23.701320Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:c1accfeec83cd81b57a393f899f7cdd6d13deb7f665cede27cc0cbb359f0a6d6","observation_id":"294d1086-4bb9-4ba4-8247-257ff3a68a55","resolution":{"observed_at":"2026-08-07T10:18:27.810127Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:27.511310Z","title":"Mooncake: A KVCache-centric disaggregated architecture for LLM serving, 2024","venue":null,"work_id":"dfcc1a96-1c52-43f1-8323-de3dd9ff5c35","year":2024},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:23.849100Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:cd595b0fa0b16b2d48ca3947c9259adf6eb8094a3ef3f442dab4805bd5455c7d","observation_id":"dd690d7d-cf99-41df-8990-1f63a9f18590","resolution":{"observed_at":"2026-08-07T10:18:27.600512Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:27.297862Z","title":"Focus: For tech giants, AI like Bing and Bard poses billion-dollar search problem","venue":null,"work_id":"c7321eb3-94e7-4fdd-8026-f213a83e6a9e","year":2023},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:23.901389Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:bc1cccadbf81ff3eba5e0424f902ff1752a3a7cea530d5ff3e3a93bca62f601d","observation_id":"fa45246a-90e3-447d-98ee-280bf8e2edb9","resolution":{"observed_at":"2026-08-07T10:18:27.412217Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:26.974952Z","title":null,"venue":null,"work_id":"c643c61b-72b9-4049-9da5-b191cf9f16e9","year":null},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:23.942676Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:e5d58a9b7319ef4fcfd75e241d700e6a016aaf941bc8c82133f0fd79ecbd6726","observation_id":"a06f4a7f-9185-4cf2-9fd8-08725cf7b59b","resolution":{"observed_at":"2026-08-07T10:18:27.145596Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:23.999552Z","title":"Megatron-lm: Training multi-billion parameter language models using model parallelism, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:23.999552Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:c999ce964bde264ed6ec0919bbc91f9ba6a15c71db1d8801853b202788a646e8","observation_id":"23909fa7-09e8-4b89-96af-7ccdb9dbc366","resolution":{"observed_at":"2026-08-07T10:18:23.999552Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:26.582950Z","title":"Gomez, Łukasz Kaiser, and Illia Polosukhin","venue":null,"work_id":"77a307ed-cfcb-48e6-9f28-713cb7826eff","year":2017},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:24.118959Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:d5a25244e6b0502b73d3df092d211dd9b16236228a91ec1d42e3510a63d13410","observation_id":"c7662fe8-a85d-4e52-a49c-f766c9634da4","resolution":{"observed_at":"2026-08-07T10:18:26.742702Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:26.279744Z","title":"https://docs.vllm.ai/en/v0.4.2/index.html","venue":null,"work_id":"94d4a7e8-f046-454c-87e6-0556b2ec68c9","year":null},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:24.229841Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:14d8a96defd816c7dd5de7c794b83aa52a910b9431173ac93278229ce4fec554","observation_id":"ed9c0fd2-6578-4842-a611-4a965f03cead","resolution":{"observed_at":"2026-08-07T10:18:26.408769Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:25.972872Z","title":"https://github.com/vllm-project/vllm-ascend","venue":null,"work_id":"7d3fcdf6-0ed2-4653-8d0e-fab3e3eb68d2","year":null},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:24.346122Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:ec051fe1b9693aad90f12ae756e17693167df6789e67a2840f1ffef002ff9e58","observation_id":"2f0c0e4d-6004-450f-8805-292f75ab23b8","resolution":{"observed_at":"2026-08-07T10:18:26.114753Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:25.738416Z","title":"Simai: Unifying architecture design and performance tuning for large-scale large language model training with scalability and precision","venue":null,"work_id":"09de92a9-b286-4f52-b79e-9a38790141c2","year":2025},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:24.449900Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:119d9691049f5a6ebfa5afaa4bbb70adb011586adc83a85d98a7e1368637ab5b","observation_id":"248364bd-e2c4-427e-ac45-4f940f6b129f","resolution":{"observed_at":"2026-08-07T10:18:25.835717Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:25.489671Z","title":"Roofline: an insightful visual performance model for multicore architectures.Commun","venue":null,"work_id":"4b495f4c-fab7-44ad-b9c7-aa7b6a00ca49","year":2009},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:24.579266Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:fa0ede0eb646382e78b57a7f75e9db8edca00bf88752dbc614baf1c8ece47c53","observation_id":"86364566-a7b8-4d13-b770-c3fc07d52163","resolution":{"observed_at":"2026-08-07T10:18:25.636873Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:25.264935Z","title":"Orca: A distributed serving system for Transformer-Based generative models","venue":null,"work_id":"de0d95b8-bc7a-43cb-a8e7-6e9b5c7a690b","year":2022},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:24.712230Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:1fbd13fc014de37eb29dda6176105071c447733a7078f8e125f930089bb11f9e","observation_id":"9ac34d5f-de27-4130-a262-bbfaf2fe842d","resolution":{"observed_at":"2026-08-07T10:18:25.376580Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:24.846042Z","title":"Root mean square layer normalization, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:24.846042Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:e542efc81370109b0aeb354a4521d7da924d53f3bb0051698faa99e2e0795756","observation_id":"9e1804fc-072c-44a9-b496-b7a853b79dec","resolution":{"observed_at":"2026-08-07T10:18:24.846042Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:25.061346Z","title":"DistServe: Disaggregating prefill and decoding for goodput-optimized large language model serving","venue":null,"work_id":"f7a39ab7-36c1-4e04-9ba4-097b00d30836","year":2024},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:24.943745Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:59694243362e8175165141946ba2239bd3196b9e9bf7964ba30cb864ef4d881e","observation_id":"a0d35056-e58c-43c2-81eb-367868a0e7fd","resolution":{"observed_at":"2026-08-07T10:18:25.146940Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-11T21:05:00.041229Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures"},"reference_resolution":{"displayed":36,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":12,"verified_exact":0,"verified_fuzzy":24},"total_outbound_references":36},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 19 August 2026, this Paper Citation Record lists 36 of 36 outbound references and 1 inbound Pith citation observation for arXiv:2506.05871."}