{"as_of":"2026-08-18T19:14:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3a4f92879cdaff2dcba0c4f73b3466fc5394dc202f2fc36007b74521f9a03106","coverage":[{"denominator":55,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":55,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T10:28:52.545392Z","state":"measured"},{"denominator":57,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":57,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-27T23:49:28.318260Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T15:27:06.035370Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"cited_work":{"arxiv_id":"2504.18154","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.18154","snapshot_observed_at":"2026-07-02T15:27:06.035370Z","title":"arXiv preprint arXiv:2504.18154 , year=","venue":null,"work_id":"9194ca84-f774-433b-8d5f-7cd9f1074f2f","year":2025},"citing_paper":{"arxiv_id":"2605.02189","last_updated":"2026-05-04T03:37:40Z","snapshot_observed_at":"2026-08-12T14:41:49.916145Z","submitted_at":"2026-05-04T03:37:40Z","title":"PipeMax: Enhancing Offline LLM Inference on Commodity GPU Servers","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-05-08T18:49:56.357400Z"},"links":{"cited_paper":"/paper/2504.18154","citing_paper":"/paper/2605.02189"},"observation_digest":"sha256:c2b984a3380c0e213fc658a8b9779e12494f78aec17ca9b9b74546af626bbba0","observation_id":"9b6e3136-9875-49ef-be5f-9f31cf4f7e5a","resolution":{"observed_at":"2026-05-09T06:10:41.552912Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"cited_work":{"arxiv_id":"2504.18154","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.18154","snapshot_observed_at":"2026-07-02T15:27:06.035370Z","title":"arXiv preprint arXiv:2504.18154 , year=","venue":null,"work_id":"9194ca84-f774-433b-8d5f-7cd9f1074f2f","year":2025},"citing_paper":{"arxiv_id":"2606.05933","last_updated":"2026-06-04T09:36:40Z","snapshot_observed_at":"2026-08-13T14:18:06.635132Z","submitted_at":"2026-06-04T09:36:40Z","title":"Beyond Greedy Chunking: SLO-Aware Sliding-Window Scheduling for LLM Inference","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-27T23:49:28.318260Z"},"links":{"cited_paper":"/paper/2504.18154","citing_paper":"/paper/2606.05933"},"observation_digest":"sha256:789c1a6c0c827896dd8a2c683c64783455db02a33f2c952e0d926a591438d447","observation_id":"d5da95d9-fc98-450b-8e17-40713bbb4b43","resolution":{"observed_at":"2026-07-02T15:27:06.037068Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2504.18154/citation-record","integrity":"/paper/2504.18154/integrity","json":"/paper/2504.18154/citation-record.json","paper":"/paper/2504.18154"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:53.070503Z","title":"Github Copilot","venue":null,"work_id":"90e69bad-3c39-431d-9383-6c5459e3f9a0","year":2023},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.343695Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:405637deb55b4a1fc89ffe7702dea7375cb4dea00494225bbca484371bf2e5bf","observation_id":"90a39ebd-a583-4439-af30-167a617cd87d","resolution":{"observed_at":"2026-08-16T10:28:53.074202Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:53.057456Z","title":null,"venue":null,"work_id":"617b357b-ef14-48f9-a0c3-0cc4171e9953","year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.348035Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:051655388dddcb2d4be208a56211cad739a99ee0f7d8788d7e91de2ea1fb60a1","observation_id":"5059a1f8-064c-42f6-a978-529a02ca2679","resolution":{"observed_at":"2026-08-16T10:28:53.062322Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:53.046889Z","title":"Faster Transformer","venue":null,"work_id":"d66b6111-c479-4998-8d0e-f975839197f7","year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.351698Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:ca80b456dcd5f0b938f9e90c0fcc7230c849810b3728e8b71c0a15b0afb8435e","observation_id":"fab712f0-b525-4fc8-bf53-3cb1b8369fd3","resolution":{"observed_at":"2026-08-16T10:28:53.050116Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:53.034612Z","title":null,"venue":null,"work_id":"058d9a09-21d7-4ff6-aaee-90f7f683b2a7","year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.357270Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:9e705ee3d73048255b90f79b3f69575ceefa724e5e5c9b5b8c842fd33b7792cd","observation_id":"1a512663-780a-4052-830e-acff74e3c9a0","resolution":{"observed_at":"2026-08-16T10:28:53.038963Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:53.021906Z","title":"vllm: Easy, fast, and cheap llm serving for everyone","venue":null,"work_id":"0ebc011f-138f-44f1-87db-fe93bc3f1329","year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.361285Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:a2664b7f1e31f48745159be87272896c0cacdefea31661b41733999ddebddc6d","observation_id":"ab5b784d-8537-453d-bd18-fe746965d4ca","resolution":{"observed_at":"2026-08-16T10:28:53.026748Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:53.008854Z","title":"Character ai","venue":null,"work_id":"c0f00bc3-e8eb-4c4b-9878-994df981f591","year":2025},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.364776Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:634dac5ac5e7e35d98c4bf66ca19ea82095f83c18befd10820870de81cd48006","observation_id":"790118fd-8436-4c32-b49c-22a6419ab4cc","resolution":{"observed_at":"2026-08-16T10:28:53.012671Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.997602Z","title":null,"venue":null,"work_id":"23da2bf4-5cf5-458e-a2a6-36fbe7512b7f","year":2025},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.369294Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:911b3697b5b08adad4ebc0d1d0754861230ae9ccf63e3b543a9af57eb33ae163","observation_id":"b45f90ec-263e-4b9c-bfb9-e73fa2baf704","resolution":{"observed_at":"2026-08-16T10:28:53.001134Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.987091Z","title":null,"venue":null,"work_id":"bb1c3c8e-de3b-4f6d-a715-8fa1ff1ee66b","year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.372385Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:dca0f8b181849436a23dc918a510ff7a5de7f79882503281cca7de40ac792f60","observation_id":"63ba6196-67ee-46b0-a317-ee3dbf7bc07d","resolution":{"observed_at":"2026-08-16T10:28:52.990574Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.974871Z","title":null,"venue":null,"work_id":"5a041ce4-1dfb-4e81-a8e3-d57f949fcef1","year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.376638Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:819687d608724371ffd4dd2a1fe2ae872adfd9808e77c07c26bbaae81fbba44f","observation_id":"31b37b73-48f4-40b5-bf81-a3b19efa2577","resolution":{"observed_at":"2026-08-16T10:28:52.979332Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.13245","last_updated":"2023-12-23T17:55:11Z","snapshot_observed_at":"2026-08-15T06:28:09.529747Z","submitted_at":"2023-05-22T17:16:38Z","title":"GQA: Training Generalized Multi-Query Transformer Models from Multi-Head Checkpoints","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.13245","snapshot_observed_at":"2026-08-16T10:28:52.380722Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.380722Z"},"links":{"cited_paper":"/paper/2305.13245","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:d748d9617265530516d84671107d3c4cc234e9df61c71c9dd1d9a75a7aed501d","observation_id":"faaf55e2-8134-4571-9111-82eb45256317","resolution":{"observed_at":"2026-08-16T10:28:52.380722Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.965670Z","title":null,"venue":null,"work_id":"bd1c02c1-7882-4e69-ba0e-bb49df3b8b57","year":2020},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.384850Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:991b29a553b7d6c3c7f3ce785eae8f8047cb22a653d7cbf242302849cceab2ca","observation_id":"b595d392-de6c-42da-b95e-379b17254809","resolution":{"observed_at":"2026-08-16T10:28:52.968749Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.388729Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.388729Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:e174e950926545106250f1ea13ee47e38e6a7e520884d06a5a02eb81a9706ef4","observation_id":"c6b5d3c4-32d2-46f3-a15e-c7a664abaa99","resolution":{"observed_at":"2026-08-16T10:28:52.388729Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.08658","last_updated":"2023-01-20T16:06:59Z","snapshot_observed_at":"2026-08-16T16:01:08.066346Z","submitted_at":"2023-01-20T16:06:59Z","title":"ATP: Adaptive Tensor Parallelism for Foundation Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.08658","snapshot_observed_at":"2026-08-16T10:28:52.392422Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.392422Z"},"links":{"cited_paper":"/paper/2301.08658","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:0ec841f30debd9747ae361841c6e2e23229fa96318a153abfd6a30c562c4a4b2","observation_id":"1591e0fc-6ccd-4224-b9d7-2d3f8e129e9c","resolution":{"observed_at":"2026-08-16T10:28:52.392422Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.397103Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.397103Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:017bac97fd589f20230aa5f7d312712b4be6a4ebd9176cbaa7e7af9c76d1fdc9","observation_id":"52a894be-3331-4b5d-b7d1-eb7742c4810f","resolution":{"observed_at":"2026-08-16T10:28:52.397103Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.937169Z","title":null,"venue":null,"work_id":"1bedf596-37cd-48c2-8953-b9e6fad83f40","year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.403130Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:61cd590aa970bc29a1ecaf60035451cda359e062e4b7ed986a668ddc4b4603ba","observation_id":"4aa76aa4-d08c-46e8-b544-cbac8bbbd9a8","resolution":{"observed_at":"2026-08-16T10:28:52.940381Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.926860Z","title":null,"venue":null,"work_id":"a0e8adbc-89df-497a-b69a-9cf44d90c82e","year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.406750Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:dd458b14e2eb547e9613351e1d33c7b43f8e3dd8464730bbccb8fb4277599542","observation_id":"4eae3516-1b02-432f-9d2d-8d83cea8777d","resolution":{"observed_at":"2026-08-16T10:28:52.929393Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.913758Z","title":null,"venue":null,"work_id":"6a6cfddb-a637-4da0-bec3-85a50df2c783","year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.409775Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:69e7c3f3e8ab2d3809891d5f6f204044abb3ba5ef8ea1cbff7b069491f709496","observation_id":"2f0a4305-3a41-4a37-b606-644bc8599afb","resolution":{"observed_at":"2026-08-16T10:28:52.918547Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-16T10:28:52.412545Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.412545Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:abfff28cc91234641159617335646596ee3f02971387deedc4e4eb0db5e38621","observation_id":"6e8ce1a6-223e-453a-9be0-50234185dc2a","resolution":{"observed_at":"2026-08-16T10:28:52.412545Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.11421","last_updated":"2024-03-18T02:30:23Z","snapshot_observed_at":"2026-08-16T14:08:58.432540Z","submitted_at":"2024-03-18T02:30:23Z","title":"FastDecode: High-Throughput GPU-Efficient LLM Serving using Heterogeneous Pipelines","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.11421","snapshot_observed_at":"2026-08-16T10:28:52.416961Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.416961Z"},"links":{"cited_paper":"/paper/2403.11421","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:b536e5ecbb874bce49f2a7817dde4bf2ed251406f0dbb818ae71d73fb67cb5ab","observation_id":"cfd21116-0766-43e2-8033-d0362687c291","resolution":{"observed_at":"2026-08-16T10:28:52.416961Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.903834Z","title":null,"venue":null,"work_id":"ffdf1f15-f396-4557-943f-b61c052f1096","year":2019},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.421192Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:672905536282c85d5463f1646e5d99d2abbfb35dd7725486366e7d5aad656508","observation_id":"21829075-7633-4d3c-af9d-45af183fbc78","resolution":{"observed_at":"2026-08-16T10:28:52.907011Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.424515Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.424515Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:562df8d759e58725859483fd4152b4f1af18a5b026af23d4797a4e3bb6b4a544","observation_id":"23fc55cc-75ad-4266-b0ae-58b5aae2ec77","resolution":{"observed_at":"2026-08-16T10:28:52.424515Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.14509","last_updated":"2023-10-04T16:51:13Z","snapshot_observed_at":"2026-08-13T22:49:56.824755Z","submitted_at":"2023-09-25T20:15:57Z","title":"DeepSpeed Ulysses: System Optimizations for Enabling Training of Extreme Long Sequence Transformer Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.14509","snapshot_observed_at":"2026-08-16T10:28:52.428447Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.428447Z"},"links":{"cited_paper":"/paper/2309.14509","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:a70c253efb7b0123d721d315a0181f9824bab01d2b55bc791bb2c0f791790e6d","observation_id":"8b828604-8365-4bd5-be17-a634221b8da6","resolution":{"observed_at":"2026-08-16T10:28:52.428447Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.12457","last_updated":"2024-04-25T06:47:57Z","snapshot_observed_at":"2026-08-18T08:35:06.332399Z","submitted_at":"2024-04-18T18:32:30Z","title":"RAGCache: Efficient Knowledge Caching for Retrieval-Augmented Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.12457","snapshot_observed_at":"2026-08-16T10:28:52.431819Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.431819Z"},"links":{"cited_paper":"/paper/2404.12457","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:f793620f426d8251c15fedd35581fc4f9f6a7a7b6d9b8b68c9eb61799a66cbf0","observation_id":"ace5662a-dee5-4e51-b39a-3b68e60323eb","resolution":{"observed_at":"2026-08-16T10:28:52.431819Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.435320Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.435320Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:822319a31a8dab066c3fcd264fee6ed4822582afa328a7e99fc15cf26ab468b9","observation_id":"8c113073-17c9-481e-8dce-b80dbd375ffe","resolution":{"observed_at":"2026-08-16T10:28:52.435320Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.874441Z","title":null,"venue":null,"work_id":"69387db0-cfb8-4ba4-b2da-f9e23740f4fb","year":2023},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.442498Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:c38a26b6100a9709cf469e91b61ddd82ab2b998711bb6020a3a9f2bbd72848d5","observation_id":"f5d524ad-90ae-4235-a3d3-843ab4fcd47e","resolution":{"observed_at":"2026-08-16T10:28:52.878056Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.864495Z","title":null,"venue":null,"work_id":"ba5c095e-5596-4c11-87ca-c2b1c18067c0","year":2021},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.445385Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:9f1f11ebe9c46324aa120a3750d85415618cc977ba2978d342769358ae04a398","observation_id":"de6751d6-02c2-4e49-bcbb-c3529214b869","resolution":{"observed_at":"2026-08-16T10:28:52.867240Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.20552","last_updated":"2025-03-26T13:48:35Z","snapshot_observed_at":"2026-08-18T09:23:54.569858Z","submitted_at":"2025-03-26T13:48:35Z","title":"Injecting Adrenaline into LLM Serving: Boosting Resource Utilization and Throughput via Attention Disaggregation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.20552","snapshot_observed_at":"2026-08-16T10:28:52.448573Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.448573Z"},"links":{"cited_paper":"/paper/2503.20552","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:3e00943ed18c2e79970e412879b5d39455cba72411e7a1775b44f04eb640016b","observation_id":"52c91951-a885-4c51-9254-d96119a2c3fc","resolution":{"observed_at":"2026-08-16T10:28:52.448573Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.02669","last_updated":"2024-07-04T15:12:54Z","snapshot_observed_at":"2026-08-17T04:58:55.259763Z","submitted_at":"2024-01-05T06:53:00Z","title":"Infinite-LLM: Efficient LLM Service for Long Context with DistAttention and Distributed KVCache","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.02669","snapshot_observed_at":"2026-08-16T10:28:52.452135Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.452135Z"},"links":{"cited_paper":"/paper/2401.02669","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:7c7dee746b57482d8f3d3aa9e647f616d309c2a05f317d39ffbf7fa9ff7c2269","observation_id":"0c40a79b-c316-4493-b732-015a2ccfaa86","resolution":{"observed_at":"2026-08-16T10:28:52.452135Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01889","last_updated":"2023-11-27T06:38:47Z","snapshot_observed_at":"2026-08-14T10:14:18.862721Z","submitted_at":"2023-10-03T08:44:50Z","title":"Ring Attention with Blockwise Transformers for Near-Infinite Context","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01889","snapshot_observed_at":"2026-08-16T10:28:52.455554Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.455554Z"},"links":{"cited_paper":"/paper/2310.01889","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:f684e0ad79553d26bf8ef41cdb6d4e4c3b713f958b2263ece213f77d29133488","observation_id":"c7480dc0-b0be-46f6-bf97-984c48aa59d4","resolution":{"observed_at":"2026-08-16T10:28:52.455554Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05821","last_updated":"2025-04-09T10:23:39Z","snapshot_observed_at":"2026-08-16T14:11:24.750510Z","submitted_at":"2024-03-09T07:01:44Z","title":"Optimizing LLM Queries in Relational Data Analytics Workloads","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05821","snapshot_observed_at":"2026-08-16T10:28:52.460076Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.460076Z"},"links":{"cited_paper":"/paper/2403.05821","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:f4b14e54f9e86c9224ff947822b242fe89e3c7f1ccb13c66d985dca3593431c4","observation_id":"eb1afff9-f975-4e09-bc0f-d9cac7d7e3a9","resolution":{"observed_at":"2026-08-16T10:28:52.460076Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.09781","last_updated":"2024-04-01T02:18:42Z","snapshot_observed_at":"2026-08-16T15:32:32.643405Z","submitted_at":"2023-05-16T20:12:59Z","title":"SpecInfer: Accelerating Generative Large Language Model Serving with Tree-based Speculative Inference and Verification","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.09781","snapshot_observed_at":"2026-08-16T10:28:52.463912Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.463912Z"},"links":{"cited_paper":"/paper/2305.09781","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:cc227e6bc2547d224df91321eb6b4b2f9bf0387dfc035d0303cb9523f107c96d","observation_id":"073825de-aa8c-4835-8b5b-32cfa0e50365","resolution":{"observed_at":"2026-08-16T10:28:52.463912Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.468739Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.468739Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:0a69c957e3664ab136a5ca78c4a71081a63c4b3761ad5ceaf09c0e73c3a7f050","observation_id":"61af3c15-2778-43f5-8733-8aca9d9d2dd0","resolution":{"observed_at":"2026-08-16T10:28:52.468739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.471742Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.471742Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:3f6e2feb7af58dc422ceda904554d2196785ae8c2cd0bbb88b103d0c79fa51d8","observation_id":"28e44035-5ae6-4119-ad1c-70d4a5d5ae17","resolution":{"observed_at":"2026-08-16T10:28:52.471742Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.04437","last_updated":"2025-01-29T04:10:41Z","snapshot_observed_at":"2026-08-18T11:18:23.884084Z","submitted_at":"2024-05-07T16:00:32Z","title":"vAttention: Dynamic Memory Management for Serving LLMs without PagedAttention","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.04437","snapshot_observed_at":"2026-08-16T10:28:52.475171Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.475171Z"},"links":{"cited_paper":"/paper/2405.04437","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:8de3f8be2a82c2f62c327dfcd1dfc46a49c9bf8b1d0ff0c81fd7fce1eb5d859c","observation_id":"a537719b-59a0-4ae9-bc80-d2e88cd0118b","resolution":{"observed_at":"2026-08-16T10:28:52.475171Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.843374Z","title":null,"venue":null,"work_id":"80b664af-cd01-4d50-ae31-e4b45ede3d4d","year":2025},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.479248Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:99267c9780c6da4c6015b35ad5d96d9dd676d41cdac6de457a75408cf99150ef","observation_id":"f39fbd49-d909-4253-8707-22dfcb49cbe4","resolution":{"observed_at":"2026-08-16T10:28:52.847119Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12950","last_updated":"2024-01-31T19:47:26Z","snapshot_observed_at":"2026-08-18T03:59:39.242039Z","submitted_at":"2023-08-24T17:39:13Z","title":"Code Llama: Open Foundation Models for Code","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12950","snapshot_observed_at":"2026-08-16T10:28:52.482825Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.482825Z"},"links":{"cited_paper":"/paper/2308.12950","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:ff22ccf16c9273e26f304e16f9daf5287ae391397387bb2d1e64f43601fd3c95","observation_id":"c2934c1c-bea9-4202-96db-0cd6f364d644","resolution":{"observed_at":"2026-08-16T10:28:52.482825Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.830530Z","title":null,"venue":null,"work_id":"8ae32bd8-ac74-43b5-8e9e-ddaae5af28f9","year":1911},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.487089Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:f5392d4e43f74aa8a1da4d42a05e8ba8a37a705aa9b982f0e47bef20ca7a4a9b","observation_id":"8633ef53-d4b1-41e2-87df-eb6b3d4b45fe","resolution":{"observed_at":"2026-08-16T10:28:52.834033Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.819118Z","title":null,"venue":null,"work_id":"82326148-c0d0-4d4b-a630-f73aedb3717d","year":2018},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.490773Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:67cfdeec363a6402d01bdaf830de4b625255aa4a24bfdff98686f60e9831d27d","observation_id":"fb1fe424-1ae6-4692-b090-bd759cb7a7c8","resolution":{"observed_at":"2026-08-16T10:28:52.822623Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.493720Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.493720Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:d20a58545c4b40b1cc743967bcc1f2f95e5f389e6be8966737befe1334907186","observation_id":"a2ef98f4-9840-4f28-b0da-176a4f042c2c","resolution":{"observed_at":"2026-08-16T10:28:52.493720Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1909.08053","last_updated":"2020-03-13T23:45:18Z","snapshot_observed_at":"2026-08-12T10:50:46.357243Z","submitted_at":"2019-09-17T19:42:54Z","title":"Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.08053","snapshot_observed_at":"2026-08-16T10:28:52.501489Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.501489Z"},"links":{"cited_paper":"/paper/1909.08053","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:50773c16d5454a8101468e64e4fc011a5fbac476f9748ee8b0fbe2bb93c632b0","observation_id":"55d5ab65-cb75-474f-abd1-6b182eeba055","resolution":{"observed_at":"2026-08-16T10:28:52.501489Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.796673Z","title":null,"venue":null,"work_id":"b4bd2c71-96ae-418f-813a-926abc8a5fdb","year":2023},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.505275Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:f9b49ff0b864352f88f74e31f219be776d9e392cec054b1f338d6ac38618e230","observation_id":"45f88ba3-0569-4aa3-bf0f-d957b6b02bd4","resolution":{"observed_at":"2026-08-16T10:28:52.799812Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.498170Z","title":"In International Conference on Machine Learning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.498170Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:6f38536da62fa77cd8c6ba21265244ea49b41e4e951e4d29398672147e654ecb","observation_id":"a24904de-6523-46b2-872a-a2dcb7bdf45e","resolution":{"observed_at":"2026-08-16T10:28:52.498170Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-16T10:28:52.512486Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.512486Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:a843c1b09c670311f4e3900cef1c5036356c796550d33f62e92da5afef357b40","observation_id":"3e56e867-a6af-4dae-abed-6347f9733b08","resolution":{"observed_at":"2026-08-16T10:28:52.512486Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.516329Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.516329Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:bb725f75ee0d3a9c799fcf4dfe8e7d00fe58d181ae8c03d860b060aef692edf7","observation_id":"340b5c5d-7843-48c0-9201-325952e0be67","resolution":{"observed_at":"2026-08-16T10:28:52.516329Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21465","last_updated":"2025-04-25T19:40:54Z","snapshot_observed_at":"2026-08-18T18:59:11.446112Z","submitted_at":"2024-10-28T19:08:12Z","title":"ShadowKV: KV Cache in Shadows for High-Throughput Long-Context LLM Inference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21465","snapshot_observed_at":"2026-08-16T10:28:52.508979Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.508979Z"},"links":{"cited_paper":"/paper/2410.21465","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:c1bb7a8a688b6a9644d1217482027ca4ef374d2256715454da5977fa9eff4554","observation_id":"d19f214f-d719-492e-83ff-27baf0ba7ae9","resolution":{"observed_at":"2026-08-16T10:28:52.508979Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10671","last_updated":"2024-09-10T13:25:53Z","snapshot_observed_at":"2026-08-17T11:08:48.802438Z","submitted_at":"2024-07-15T12:35:42Z","title":"Qwen2 Technical Report","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10671","snapshot_observed_at":"2026-08-16T10:28:52.523540Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.523540Z"},"links":{"cited_paper":"/paper/2407.10671","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:c830f2b98592f84241f2ba9608eb585fbcb463d071de7683826e773439aeecff","observation_id":"3412e589-6bb7-447f-993c-78c4968a6340","resolution":{"observed_at":"2026-08-16T10:28:52.523540Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.527545Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.527545Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:504c0667a07025a26c4e3d99a7961b508791d174e377fa964d22c5dc86a517df","observation_id":"61225a82-7ee6-43b7-b7ff-3bd027395a13","resolution":{"observed_at":"2026-08-16T10:28:52.527545Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.780452Z","title":null,"venue":null,"work_id":"a66d193d-64c9-4aca-b204-8b7ff3dbcbbf","year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.520287Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:58057292628d1f9c19017c473098d337c300775de090ffdb3fb40a9ab0680973","observation_id":"38ee51ee-5e49-42de-9c2d-f94d75adb60e","resolution":{"observed_at":"2026-08-16T10:28:52.784003Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.534148Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.534148Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:5ed4559b0073fca66253cd588c1d896ae5d4cb13a1a8f9378d9c012f625e2383","observation_id":"14986e58-8f2a-4aa0-bca2-c52300b3fac4","resolution":{"observed_at":"2026-08-16T10:28:52.534148Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.746001Z","title":"2024.{DistServe}: Disaggregating Prefill and Decoding for Goodput-optimized Large Language Model Serving","venue":null,"work_id":"778201bf-55b1-4911-84a6-014d637e1915","year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.537827Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:5bc5255301784fd944a537964d52a10af011297f35cc64bcc12db1ee32739fd5","observation_id":"74e009ea-b3cc-421a-815d-4dae67bb665f","resolution":{"observed_at":"2026-08-16T10:28:52.750772Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.763262Z","title":null,"venue":null,"work_id":"09a7bc77-3dd1-48c6-8833-132966fd2dad","year":2022},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.531054Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:d8b83983a5714e1b3804984ae5a0a67dfdb2c9ded14c3840c7690a479ecea5ea","observation_id":"38676f8d-7ba9-4aed-940a-ffd7b138c995","resolution":{"observed_at":"2026-08-16T10:28:52.766731Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.02263","last_updated":"2025-07-26T15:29:10Z","snapshot_observed_at":"2026-08-16T12:44:24.037349Z","submitted_at":"2025-04-03T04:20:44Z","title":"MegaScale-Infer: Serving Mixture-of-Experts at Scale with Disaggregated Expert Parallelism","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.02263","snapshot_observed_at":"2026-08-16T10:28:52.545392Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.545392Z"},"links":{"cited_paper":"/paper/2504.02263","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:8237943c49159de474e0f1d7e50ede2c23c1af0f8b5622b70c14407285d3ca1d","observation_id":"d148bb34-cbcb-481e-ba96-560cfb75257d","resolution":{"observed_at":"2026-08-16T10:28:52.545392Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12757","last_updated":"2025-05-25T14:08:01Z","snapshot_observed_at":"2026-08-17T17:38:19.809822Z","submitted_at":"2024-08-22T23:00:40Z","title":"NanoFlow: Towards Optimal Large Language Model Serving Throughput","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12757","snapshot_observed_at":"2026-08-16T10:28:52.541781Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.541781Z"},"links":{"cited_paper":"/paper/2408.12757","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:30db0d601f6f060ea6c5bd46b882603c672f50e7edef4b8e5c52163e1467bd9e","observation_id":"a72c7eef-a6da-43ce-8259-320b0f05cb75","resolution":{"observed_at":"2026-08-16T10:28:52.541781Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.400146Z","title":"Advances in neural information processing systems 35 (2022), 16344–16359","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.400146Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:29b0295891b09d51742ce8fa4776f1c6c67edd499a17f6c3524971a2f766addc","observation_id":"f5c06792-eb29-4484-a705-8ed01eded5bf","resolution":{"observed_at":"2026-08-16T10:28:52.400146Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.439069Z","title":"In Proceedings of the 29th Symposium on Operating Systems Principles","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.439069Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:6376f709e895e5dcb59053945ccf884448d343346f683a2fb75899d91d52e065","observation_id":"a64b634a-0847-4491-83e8-2d4cd932fee9","resolution":{"observed_at":"2026-08-16T10:28:52.439069Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","latest_version":1,"primary_category":"cs.DC","snapshot_observed_at":"2026-08-18T18:55:10.997004Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration"},"reference_resolution":{"displayed":55,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":50,"verified_exact":0,"verified_fuzzy":5},"total_outbound_references":55},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 55 of 55 outbound references and 2 inbound Pith citation observations for arXiv:2504.18154."}