{"as_of":"2026-08-03T16:39:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:955859407ebf13bdcb84eee81df4e85fe06ea85beb9642c9ab839b9d0fb1db38","coverage":[{"denominator":44,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":44,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-22T10:20:45.375375Z","state":"measured"},{"denominator":44,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":44,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-03T06:30:56.289259+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2605.07985/citation-record","integrity":"/paper/2605.07985/integrity","json":"/paper/2605.07985/citation-record.json","paper":"/paper/2605.07985"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"URLhttps://developer.nvidia.com/cupti","venue":null,"work_id":"bd9bffcf-debd-429f-8f53-9512a84a45ba","year":2025},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:daae5fd5c406417c6457fba3e80417861fa318506e296e2b9be5293715d450e3","observation_id":"f5328f67-d8a0-4e06-864c-6b3dfa09ab1f","resolution":{"observed_at":"2026-05-22T10:21:25.200462Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"URL https://docs.pytorch.org/tutorials/ recipes/recipes/profiler_recipe.html","venue":null,"work_id":"eceee1c8-c845-4e18-9100-3f03ed64c093","year":2026},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:8831c66cbb1ed437b6dbeb669262aaad883131e1209be5fa2ba41507b1942e06","observation_id":"a929d66d-0f32-474b-a411-6b97ecb62280","resolution":{"observed_at":"2026-05-22T10:21:25.204791Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.05465","last_updated":"2024-05-21T05:17:29Z","snapshot_observed_at":"2026-07-06T18:11:53.063337Z","submitted_at":"2024-05-08T23:42:13Z","title":"Vidur: A Large-Scale Simulation Framework For LLM Inference","version":2},"cited_work":{"arxiv_id":"2405.05465","doi":null,"metadata_source":"pith","pith_arxiv_id":"2405.05465","snapshot_observed_at":"2026-07-11T01:57:50.732280Z","title":"Vidur: A large-scale simulation framework for llm inference","venue":"cs.LG","work_id":"c6236eea-697f-4c84-8ad3-41279c46d8ee","year":2024},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"cited_paper":"/paper/2405.05465","citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:36040db0937ac1567f9161648a25bcb411717454541daf2ab876998003b48e96","observation_id":"0c203058-aa1f-4c39-83db-a888fd5c1316","resolution":{"observed_at":"2026-05-22T10:21:23.846209Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":"2403.02310","doi":"10.48550/arxiv.2403.02310","metadata_source":"pith","pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-07-11T01:57:51.352536Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":"cs.LG","work_id":"07bbeba4-bd20-4699-9af8-2d24435f090a","year":2024},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:aded49c154c88bd1db993814251fb6f93cde61784ae55d24a2cf74ceed8aa80a","observation_id":"74d58261-838b-4fd0-808d-b7d5c2fa05f4","resolution":{"observed_at":"2026-05-22T10:21:23.908790Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2601.00397","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Revati: Transparent gpu-free time-warp emulation for llm serving","venue":null,"work_id":"4ad8f1f1-619c-43e2-b7a7-e90138bed226","year":2026},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:70f217524c493950cf1e66d698a521bbcdfb8ed155fbb0fed9588bc421a008b6","observation_id":"3172fe24-8cd0-41fc-a150-7bc8a21bc2ac","resolution":{"observed_at":"2026-05-22T10:21:23.885158Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.13245","last_updated":"2023-12-23T17:55:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-22T17:16:38Z","title":"GQA: Training Generalized Multi-Query Transformer Models from Multi-Head Checkpoints","version":3},"cited_work":{"arxiv_id":"2305.13245","doi":"10.48550/arxiv.2305.13245","metadata_source":"pith","pith_arxiv_id":"2305.13245","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"GQA: Training Generalized Multi-Query Transformer Models from Multi-Head Checkpoints","venue":"cs.CL","work_id":"b73ad5b2-e553-4c71-b0c9-67e67ba7b158","year":2023},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"cited_paper":"/paper/2305.13245","citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:6422eb7e61e3a43372ed12688866f99a653592bd6e4ead4eb93337a57897e886","observation_id":"ce950114-2d3f-44af-9e51-84605f9f3e20","resolution":{"observed_at":"2026-05-22T10:21:23.864756Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"vllm v1 performance optimization","venue":null,"work_id":"969c22bf-897e-43d3-a610-c4359ea4a906","year":2026},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:3f26d179f221fae29e3c82e22957082c3344b3e8fb0b8f854b615e84fd838f89","observation_id":"054f693d-0569-4103-9b37-eaec98632265","resolution":{"observed_at":"2026-05-22T10:21:25.191099Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"4291.25942","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Flowdroid: precise context, flow, field, object-sensitive and lifecycle-aware taint analysis for android apps","venue":null,"work_id":"835984bc-c6c0-41b9-b24c-1a2c81ae9d8d","year":2014},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:de0c35c087c0eae54aab86dba1ab1e371e2b925ed35f3988c716a94d9935e62c","observation_id":"009ae2dc-da17-41e9-a1a6-1da67c424bd6","resolution":{"observed_at":"2026-05-22T10:21:23.902334Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Peters, and Arman Cohan","venue":null,"work_id":"8e3e2706-5fa0-4f26-a91a-e0d45bb9d08d","year":null},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:003795bf067d2aaef6dc42035df3e559214457ed20ecb2123b918b46a9789b4f","observation_id":"dc7372b2-dfb7-4258-ab09-e0ae8bbe5f08","resolution":{"observed_at":"2026-05-22T10:21:25.195664Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2004.05150","last_updated":"2020-12-02T17:52:35Z","snapshot_observed_at":"2026-07-31T17:17:17.205582Z","submitted_at":"2020-04-10T17:54:09Z","title":"Longformer: The Long-Document Transformer","version":2},"cited_work":{"arxiv_id":"2004.05150","doi":"10.48550/arxiv.2004.05150","metadata_source":"pith","pith_arxiv_id":"2004.05150","snapshot_observed_at":"2026-07-11T03:27:46.681766Z","title":"Longformer: The Long-Document Transformer","venue":"cs.CL","work_id":"abea7a44-6668-4de7-aab6-f53a6e5aa088","year":2020},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"cited_paper":"/paper/2004.05150","citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:2d30484d317b15f0a2ee69201637368b226b9c700da64364bfa419e0a0c408ff","observation_id":"759531f3-8c17-40a9-b079-d9c45eb24908","resolution":{"observed_at":"2026-05-22T10:21:23.859081Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-07-12T21:49:59.161233+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T21:49:59.161233+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.08784","last_updated":"2025-04-05T17:41:26Z","snapshot_observed_at":"2026-07-06T21:08:05.749656Z","submitted_at":"2025-04-05T17:41:26Z","title":"SLOs-Serve: Optimized Serving of Multi-SLO LLMs","version":1},"cited_work":{"arxiv_id":"2504.08784","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.08784","snapshot_observed_at":"2026-07-10T05:16:47.916233Z","title":"Slos-serve: Optimized serving of multi-slo llms","venue":"cs.DC","work_id":"8a37827e-843e-4c6a-ba87-6f24004bf54b","year":2025},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"cited_paper":"/paper/2504.08784","citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:c8ee225c57f79f8b7c8a0d1991f7e4c6242b165acd3f217d0858b6dee9fc123a","observation_id":"4fe38901-838a-4ce5-ad65-c01ac960eeee","resolution":{"observed_at":"2026-05-22T10:21:23.920717Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"3097.2024","doi":"10.1109/iiswc63097.2024.00012","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Llmservingsim: A hw/sw co-simulation infrastructure for llm inference serving at scale","venue":null,"work_id":"56fc806b-3fa8-47de-a622-00d820ed3b49","year":2024},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:00ae71ae817c19c0cfbb8664ddaa35f16fda6443e32124dced7c7c77e6a301ed","observation_id":"22b3bfe9-c3c8-4d82-b19f-a8621e3f6d3f","resolution":{"observed_at":"2026-05-22T10:21:23.964247Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-05-23T16:53:08.722415+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T16:53:08.722415+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2025.362832","doi":"10.1109/lca.2025.3628325","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Llmservingsim2.0: A unified simulator for het- erogeneous hardware and serving techniques in llm infrastructure.IEEE Computer Architecture Letters, 24(2):361–364, July 2025","venue":null,"work_id":"895064f6-3f81-4498-be9b-9167237e1ef8","year":2025},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:5a3b092048e6daf8b630c78649e3347436e30247761907592f7460ad48b498b2","observation_id":"60463572-b9f7-4e42-9a0d-0fddb59fd3d3","resolution":{"observed_at":"2026-05-22T10:21:23.148976Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Coherelabs/c4ai-command-r7b-12-2024","venue":null,"work_id":"c13061bf-25f5-4b7d-8155-5bc2085422b0","year":2024},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:0bcde0a4b6f81e7636297bd65692e6ababc079d4f330e3960cc43485aecb87ee","observation_id":"a7a66680-9994-4ab0-b226-e9e213cb857d","resolution":{"observed_at":"2026-05-22T10:21:25.209651Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.14135","last_updated":"2022-06-23T17:53:32Z","snapshot_observed_at":"2026-07-06T13:14:48.753329Z","submitted_at":"2022-05-27T17:53:09Z","title":"FlashAttention: Fast and Memory-Efficient Exact Attention with IO-Awareness","version":2},"cited_work":{"arxiv_id":"2205.14135","doi":"10.48550/arxiv.2205.14135","metadata_source":"pith","pith_arxiv_id":"2205.14135","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"FlashAttention: Fast and Memory-Efficient Exact Attention with IO-Awareness","venue":"cs.LG","work_id":"efa96825-0830-4cfc-a250-fdaf6af302ab","year":2022},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"cited_paper":"/paper/2205.14135","citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:d9cacbc8cef907d5fb4ee6ccb3875c0da165a4e8a8a11db20a1cd029f936a9d7","observation_id":"de0e2621-d09f-4f90-8c66-fa4f8c735269","resolution":{"observed_at":"2026-05-22T10:21:23.870869Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-07-13T15:50:18.534404+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-13T15:50:18.534404+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Adapting vidur to vllm and profiling cpu overhead","venue":null,"work_id":"d78fbff1-42c2-473b-a517-29fa797873a5","year":2025},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:69e8186e449bd9246109fc1220881d8aff244ab8a8e2fbd87b8481f33e7a177a","observation_id":"74f33378-fdee-438e-9aa2-d1f67177a19e","resolution":{"observed_at":"2026-05-22T10:21:25.213995Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Cox, Jaeyeon Jung, Patrick McDaniel, and Anmol N","venue":null,"work_id":"12af5856-b35a-4142-b729-4338b9dc1bb3","year":2014},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:4c200d66bf3a2bf86a59043249ad5da2f940ad2558166bdce4329941b5ff7dac","observation_id":"d0f0d187-be0d-4e01-8178-e466e244c53a","resolution":{"observed_at":"2026-05-22T10:21:25.218224Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.03148","last_updated":"2025-08-05T06:53:28Z","snapshot_observed_at":"2026-07-06T22:07:59.033674Z","submitted_at":"2025-08-05T06:53:28Z","title":"Frontier: Simulating the Next Generation of LLM Inference Systems","version":1},"cited_work":{"arxiv_id":"2508.03148","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.03148","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Frontier: Simulating the next generation of llm inference systems","venue":null,"work_id":"d5c5ad2c-a443-47e6-9cfd-6b13a1da2d18","year":2025},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"cited_paper":"/paper/2508.03148","citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:809c515e5a851ef824ccc97f327fd679bf646c328e73c56b01b2deb79cd85322","observation_id":"60ed9ba8-68f0-44c7-b38b-8beeeb135190","resolution":{"observed_at":"2026-05-22T10:21:23.877745Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"How to get the profile csv of vllm instead of tensorrt-llm? https://github.com/ casys-kaist/LLMServingSim/issues/10","venue":null,"work_id":"21d2cfc1-2227-45e0-9e69-c87285a9a586","year":2025},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:a526d811feb3548be7e902e114b1c44d791f3db0c1ddb5886a4346fa6cdc8a17","observation_id":"c6a3ee4c-8928-4d87-bf0a-d33c8bf58146","resolution":{"observed_at":"2026-05-22T10:21:25.222822Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Perfetto trace viewer","venue":null,"work_id":"6656c0b0-457a-433b-a931-71ca26f9af64","year":null},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:e06986b7cbf2db02f2eb5480d1ee609f0eaf50ccf840c3c61d62029261d21bd5","observation_id":"cbd2f16b-0585-420d-92b4-13ca88adb0fd","resolution":{"observed_at":"2026-05-22T10:21:25.260581Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Questions about simulation fidelity under vllm version differences, profiling utiliza- tion, and throughput estimation","venue":null,"work_id":"8a6fd0ac-9a0d-4483-9f8d-d0b3acf60aac","year":2025},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:f2d2f918375e8cd2a82d2823e39558507c6ffdb8475690e8a20329f04f43ed94","observation_id":"59afc0ab-bdc2-4fc5-8d89-c6a908422a7f","resolution":{"observed_at":"2026-05-22T10:21:25.187052Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.04088","last_updated":"2024-01-08T18:47:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-01-08T18:47:34Z","title":"Mixtral of Experts","version":1},"cited_work":{"arxiv_id":"2401.04088","doi":"10.48550/arxiv.2401.04088","metadata_source":"pith","pith_arxiv_id":"2401.04088","snapshot_observed_at":"2026-07-11T03:27:47.028679Z","title":"Mixtral of Experts","venue":"cs.LG","work_id":"0de8c352-9daa-4e1e-8c7b-3d0dec69f369","year":2024},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"cited_paper":"/paper/2401.04088","citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:12f260df465a3c81911159a31d725956c804b93d5e4d88b89e62093b4fe012eb","observation_id":"b5e94dda-f203-4f76-82c9-bf1b7312c9b4","resolution":{"observed_at":"2026-05-22T10:21:23.840418Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-07-09T08:48:39.110013+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T08:48:39.110013+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"6752.348006","doi":"10.1145/3466752.3480063","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Rogers, Tor M","venue":null,"work_id":"dd6db725-af99-47bc-97ea-dc9a90a98866","year":2021},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:bef1abb79993a11cd769c5b0122eeac5b85d1769e9c2c6c10b06b54c66b156c6","observation_id":"da79cacd-db46-4382-b608-2f0a7c5d7d18","resolution":{"observed_at":"2026-05-22T10:21:23.137659Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T18:24:01.262844Z","title":"Gonzalez, Hao Zhang, and Ion Stoica","venue":null,"work_id":"ba38df18-25de-4771-8903-3ff55d99b09f","year":2023},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:92449becdfd27951f3eea697dd550bf4eb1bc4e60adf5ffb179a3c0744e63001","observation_id":"aa6391b6-7c2a-41f8-b8a5-18debb4bd7c1","resolution":{"observed_at":"2026-05-22T10:21:25.168308Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.17651","last_updated":"2025-04-29T20:48:45Z","snapshot_observed_at":"2026-07-06T19:57:28.746485Z","submitted_at":"2024-11-26T18:16:56Z","title":"APEX: An Extensible and Dynamism-Aware Simulator for Automated Parallel Execution in LLM Serving","version":2},"cited_work":{"arxiv_id":"2411.17651","doi":"10.48550/arxiv.2411.17651","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.17651","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Apex: An extensible and dynamism-aware simulator for automated parallel execution in llm serving","venue":null,"work_id":"d513a047-ee41-48d4-a6e0-4f45323f6cde","year":2025},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"cited_paper":"/paper/2411.17651","citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:3300b5be2f0fb083c629eece4fb4b28b8359df4fb8c5e38a41e5f84ec3d3c198","observation_id":"f6ef9946-1994-40d8-90ad-b6335be2eaa4","resolution":{"observed_at":"2026-05-22T10:21:23.893409Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"meta-llama/llama-3.1-8b.https://huggingface.co/meta-llama/Llama-3.1-8B","venue":null,"work_id":"04438e9f-7d3a-4450-bca3-f935a1f39335","year":null},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:567d75f9fa9be67a7688b2956b6722e2644a0370cb5b186bb4f420bff0f0dc6e","observation_id":"f4e5065c-2bac-4ad8-9201-4cc59a83294a","resolution":{"observed_at":"2026-05-22T10:21:25.177822Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Dynamic taint analysis for automatic detection, analysis, and signature generation of exploits on commodity software","venue":null,"work_id":"9c6ca02a-4d2b-4593-b70b-04f9b6e5f997","year":2005},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:7513b789d818195702884e8c13f5fb094417205d7c319db0ac89109594779134","observation_id":"774b07b9-a065-43ab-b5fa-a1576ea5ca40","resolution":{"observed_at":"2026-05-22T10:21:25.172849Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"openchat_sharegpt4_dataset","venue":null,"work_id":"688b9253-4b4a-40e3-94e8-c0c6a7d0bd2f","year":null},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:f0ff02848dfb5aefd1f0e88d9100ca83b17773267348a2f596eb71b3d10cbc20","observation_id":"51342f58-94ba-492b-b4e9-78d093efed26","resolution":{"observed_at":"2026-05-22T10:21:25.182570Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Pytorch: An imperative style, high-performance deep learning library","venue":null,"work_id":"9738a540-5d04-4973-922f-3378b09e67c3","year":2019},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:a9cb2bfff43a58ef26351a470123d0c50f6164f917ff5e848504af09eb3ef501","observation_id":"3810615a-248e-44f5-9d7b-af8b433f42a8","resolution":{"observed_at":"2026-05-22T10:21:25.265028Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.18677","last_updated":"2024-05-20T15:37:36Z","snapshot_observed_at":"2026-07-06T16:55:05.210009Z","submitted_at":"2023-11-30T16:24:42Z","title":"Splitwise: Efficient generative LLM inference using phase splitting","version":2},"cited_work":{"arxiv_id":"2311.18677","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.18677","snapshot_observed_at":"2026-07-11T01:57:51.633849Z","title":"Splitwise: Efficient generative llm inference using phase splitting","venue":"cs.AR","work_id":"ce7902cc-fd2e-4179-b84b-1f819931e7ae","year":2023},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"cited_paper":"/paper/2311.18677","citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:7aae5a5079386722780f648bba540cd336a80df1780d949ab3599cbbba4fd215","observation_id":"34cec221-9895-4ea2-93b7-9fb19a6ad154","resolution":{"observed_at":"2026-05-22T10:21:23.951734Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"GitHub - pytorch/kineto: A CPU+GPU Profiling library that provides access to timeline traces and hardware performance counters","venue":null,"work_id":"821411f2-5eda-45a3-8a4e-a7e1e79eb10a","year":null},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:481b89715fe3dffd9d37c5ba38406582854512b21646057b7c8ec6c862292d42","observation_id":"0a374a57-3407-4c84-b848-d93fc40c6062","resolution":{"observed_at":"2026-05-22T10:21:25.250511Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.11581","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T05:59:37.994563Z","title":"The anatomy of a triton attention kernel","venue":null,"work_id":"a4e51971-3f69-4198-bfd1-2aef48b463ae","year":2025},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:0af06d48eb803991b8702f9d5bd9d82ecb5aa9f9357ad7d51b869b7319927702","observation_id":"cb5325ed-7aba-4b7e-b1e1-cacff01c5ecf","resolution":{"observed_at":"2026-05-22T10:21:23.957706Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Running problems with replica scheduler orca/sarathi","venue":null,"work_id":"1e6f0dcb-297c-4d66-a848-e307f14097db","year":2025},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:6a613a91af90734c3af61d84b65b63c726a20eb8d4561dddd306db0e5f6aeb79","observation_id":"12447c86-b130-4eeb-9c69-c8ff35a2932c","resolution":{"observed_at":"2026-05-22T10:21:25.255340Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Adding new model","venue":null,"work_id":"64c60916-37f4-41e8-95f5-b6964b708911","year":2024},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:d93ba45e7fd1029d479007c97ab0e6e2424ba1416bd78465b123a667339b97f1","observation_id":"415acae4-fab2-44f2-a70c-3ee4b3f0b4ac","resolution":{"observed_at":"2026-05-22T10:21:25.241485Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1701.06538","last_updated":"2017-01-23T18:10:00Z","snapshot_observed_at":"2026-07-06T05:27:13.416519Z","submitted_at":"2017-01-23T18:10:00Z","title":"Outrageously Large Neural Networks: The Sparsely-Gated Mixture-of-Experts Layer","version":1},"cited_work":{"arxiv_id":"1701.06538","doi":"10.48550/arxiv.1701.06538","metadata_source":"pith","pith_arxiv_id":"1701.06538","snapshot_observed_at":"2026-07-11T01:07:45.051014Z","title":"Outrageously Large Neural Networks: The Sparsely-Gated Mixture-of-Experts Layer","venue":"cs.LG","work_id":"2c6b3f6d-54e4-4df7-baa7-475a490799af","year":2017},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"cited_paper":"/paper/1701.06538","citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:b46a5f217e23fc01d23f1c1fc782dd32f5b9342eb4b88cf6dde2da433adc23b1","observation_id":"379edae6-da17-4c29-9bf4-80a2e541f916","resolution":{"observed_at":"2026-05-22T10:21:23.933069Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-01T08:08:24.174744+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T08:08:24.174744+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Trtion.https://triton-lang.org/main/index.html","venue":null,"work_id":"c8771079-b56e-4de5-9e24-f17c1566a39f","year":2020},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:4d14365fcb13ab04dc4fba1fe4fdcfc83a944ef30f0db5ed9c20b96c86d42279","observation_id":"2fd53c4c-aa90-4ec1-91a5-8556d7a4a344","resolution":{"observed_at":"2026-05-22T10:21:25.245947Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1706.03762","last_updated":"2023-08-02T00:41:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-06-12T17:57:34Z","title":"Attention Is All You Need","version":7},"cited_work":{"arxiv_id":"1706.03762","doi":"10.1186/s13550-021-00830-6","metadata_source":"pith","pith_arxiv_id":"1706.03762","snapshot_observed_at":"2026-07-14T22:52:12.858825Z","title":"Attention Is All You Need","venue":"cs.CL","work_id":"baafb5a2-5272-43bc-932f-09fa9ffe5316","year":2017},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"cited_paper":"/paper/1706.03762","citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:7332b8a3e44bd59094fdcf8a730e730d4cb4e51fdf601054601629692018c2a9","observation_id":"96741a9e-a8f9-4b6f-ba03-56285caffc49","resolution":{"observed_at":"2026-05-22T10:21:23.914260Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Optimization and tuning","venue":null,"work_id":"3baef20b-1161-489b-be3e-eb7b0dd67dc3","year":2026},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:1f2954e95fc2e6b75c7728e066cdde06890fb3960f4bc7c31da2d6a268a7991e","observation_id":"afdd0979-a995-4960-854d-45b980151840","resolution":{"observed_at":"2026-05-22T10:21:25.236958Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15383","last_updated":"2025-01-26T03:47:25Z","snapshot_observed_at":"2026-07-31T01:49:29.562668Z","submitted_at":"2025-01-26T03:47:25Z","title":"Qwen2.5-1M Technical Report","version":1},"cited_work":{"arxiv_id":"2501.15383","doi":"10.48550/arxiv.2501.15383","metadata_source":"pith","pith_arxiv_id":"2501.15383","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Qwen2.5-1M Technical Report","venue":"cs.CL","work_id":"397e3b66-80bd-403f-bcf9-9a907eed1830","year":2025},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"cited_paper":"/paper/2501.15383","citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:ee70f2d621ba479572145c00c1fb2c92f238450c639de94f0be3dfdcd96e75fe","observation_id":"be340fb5-53e1-4131-a294-6e289c2d1de8","resolution":{"observed_at":"2026-05-22T10:21:23.926791Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01005","last_updated":"2025-04-21T20:10:11Z","snapshot_observed_at":"2026-07-06T20:15:36.280948Z","submitted_at":"2025-01-02T02:02:20Z","title":"FlashInfer: Efficient and Customizable Attention Engine for LLM Inference Serving","version":2},"cited_work":{"arxiv_id":"2501.01005","doi":"10.48550/arxiv.2501.01005","metadata_source":"pith","pith_arxiv_id":"2501.01005","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"FlashInfer: Efficient and Customizable Attention Engine for LLM Inference Serving","venue":"cs.DC","work_id":"a153885b-2460-4177-9053-8d0011adfcb9","year":2025},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"cited_paper":"/paper/2501.01005","citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:54df731040ede3acf759f1fb3a30d6d35ca40752e07fd56d240051a5826b08c4","observation_id":"7d2b2aba-b195-40a6-b041-97eb21c9455a","resolution":{"observed_at":"2026-05-22T10:21:23.939335Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.07104","last_updated":"2024-06-06T00:10:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-12T09:34:27Z","title":"SGLang: Efficient Execution of Structured Language Model Programs","version":2},"cited_work":{"arxiv_id":"2312.07104","doi":"10.48550/arxiv.2312.07104","metadata_source":"pith","pith_arxiv_id":"2312.07104","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"SGLang: Efficient Execution of Structured Language Model Programs","venue":"cs.AI","work_id":"bcc79aed-93ef-4b91-94e6-e62b5fa15ce7","year":2023},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"cited_paper":"/paper/2312.07104","citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:797854a7f4d0995d5d980cdf0166a9c55a29240e9fe0ccad5441f1c402e48145","observation_id":"28060f05-4488-40bc-9fcd-36cec22f71c6","resolution":{"observed_at":"2026-05-22T10:21:23.945417Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.09670","last_updated":"2024-06-06T15:50:51Z","snapshot_observed_at":"2026-08-02T14:35:40.250937Z","submitted_at":"2024-01-18T01:03:38Z","title":"DistServe: Disaggregating Prefill and Decoding for Goodput-optimized Large Language Model Serving","version":3},"cited_work":{"arxiv_id":"2401.09670","doi":"10.48550/arxiv.2401.09670","metadata_source":"pith","pith_arxiv_id":"2401.09670","snapshot_observed_at":"2026-07-11T01:57:50.833426Z","title":"Distserve: Disaggregating prefill and decoding for goodput-optimized large language model serving","venue":"cs.DC","work_id":"04db917c-d1d5-44e1-a54e-280c2f19b98e","year":2024},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"cited_paper":"/paper/2401.09670","citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:ba1fcbfeac2c4bd2e6b223b3571d5e6906f8814534e2c11d944b0876c61c0961","observation_id":"3d128a9a-6b6c-4f2b-90a8-65647c752500","resolution":{"observed_at":"2026-05-22T10:21:23.852947Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"cda4e74e-f58e-416a-8038-cb0813d463f7","year":null},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:3865d557178bac0f436b246f7ea858d10799ee089e3a4625dce4703a2df1f40d","observation_id":"1052ea2a-b780-4a90-be3a-6f1635976c59","resolution":{"observed_at":"2026-05-22T10:21:25.227519Z","resolver_source":"raw_fallback","status":"parse_uncertain"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Sufficiency of a single trace.A natural concern is whether a single dummy-prompt trace can cover both prefill and decode call paths","venue":null,"work_id":"a72622d2-258b-4055-ae6a-01cca6e04c2e","year":2024},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:06c527548868c6dd1a467ccd35540b7acc788f3a06961d37ba1b2321c3cea2c5","observation_id":"ef50f773-b9dd-448d-924a-697ffae34778","resolution":{"observed_at":"2026-05-22T10:21:25.232247Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","latest_version":2,"primary_category":"cs.DC","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation"},"reference_resolution":{"displayed":44,"state_counts":{"malformed_identifier":4,"metadata_mismatch":2,"parse_uncertain":1,"unresolved":0,"verified_exact":18,"verified_fuzzy":19},"total_outbound_references":44},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"thesis":"As of 3 August 2026, this Paper Citation Record lists 44 of 44 outbound references and 0 inbound Pith citation observations for arXiv:2605.07985."}