{"as_of":"2026-08-19T15:47:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:513d0dfb49b2e1d7d2f00a169a9b522452db3c24bfc7bd3441fbba658ed4494a","coverage":[{"denominator":42,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":42,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T18:28:01.641782Z","state":"measured"},{"denominator":42,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":42,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2411.11560/citation-record","integrity":"/paper/2411.11560/integrity","json":"/paper/2411.11560/citation-record.json","paper":"/paper/2411.11560"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:01.428717Z","title":"A survey on large language models: Applications, challenges, limitations, and practical usage","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.428717Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:23cf97ca1f5ee02d42c6ce275aa66ac0b4da8e043c073f8795deb28e36f6ece7","observation_id":"8f64c548-f65d-4eba-b4dd-645c130c0fbc","resolution":{"observed_at":"2026-08-12T18:28:01.428717Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.20018","last_updated":"2024-07-29T13:53:27Z","snapshot_observed_at":"2026-08-19T00:33:35.257508Z","submitted_at":"2024-07-29T13:53:27Z","title":"Efficient Training of Large Language Models on Distributed Infrastructures: A Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.20018","snapshot_observed_at":"2026-08-12T18:28:01.433698Z","title":"Efficient training of large language models on distributed infrastructures: A survey","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.433698Z"},"links":{"cited_paper":"/paper/2407.20018","citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:793a895e412839bff6de8a3d8cefd8b70b055a8c6790fb42a98f62cb6d6c2b29","observation_id":"b3303931-8436-4e1c-b87e-27fcafcf2e7a","resolution":{"observed_at":"2026-08-12T18:28:01.433698Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.17644","last_updated":"2025-05-26T16:16:43Z","snapshot_observed_at":"2026-08-16T14:36:00.835576Z","submitted_at":"2024-01-31T07:52:48Z","title":"BurstGPT: A Real-world Workload Dataset to Optimize LLM Serving Systems","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.17644","snapshot_observed_at":"2026-08-12T18:28:01.438721Z","title":"Towards efficient and reliable llm serving: A real-world workload study","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.438721Z"},"links":{"cited_paper":"/paper/2401.17644","citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:243afe39b665eac4cc9218eec4b238509b93909eedba660a0e3f940198fcf365","observation_id":"decb8d3c-e512-4ab7-a628-a57eb2b26359","resolution":{"observed_at":"2026-08-12T18:28:01.438721Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:02.301897Z","title":"Gödel: Unified large-scale resource management and scheduling at bytedance","venue":null,"work_id":"4a82d043-5917-4e2d-9172-163c371c8009","year":2023},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.443368Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:e814cb4a13e5647a244eac026b5004b35a6e575bfdf00420af3340b26c3758be","observation_id":"1838389c-2088-4752-ae94-a643ce2ca9f1","resolution":{"observed_at":"2026-08-12T18:28:02.306668Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:02.285958Z","title":"Topology-aware gpu scheduling for learning workloads in cloud environments","venue":null,"work_id":"8887ccee-9027-426c-92fe-519e596e72aa","year":2017},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.447617Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:743237015e2df4dd03d7ddc12e501814329ecd2e1bab6e43a35307a330ca78d3","observation_id":"686d5a79-898d-46f3-9724-e2a7a3495913","resolution":{"observed_at":"2026-08-12T18:28:02.291237Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:02.270623Z","title":"Numa (non-uniform memory access): An overview: Numa becomes more common because memory controllers get close to execution units on microprocessors","venue":null,"work_id":"97d91b2b-b0a3-4b0e-b0fd-48499972a0b8","year":2013},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.452715Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:a368b327ae167b01a118d65a53fbaea2bc5201b83248454667449996ab9c6590","observation_id":"cb7af90d-5efa-4522-b89b-27aadcf7a300","resolution":{"observed_at":"2026-08-12T18:28:02.275245Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.14527","last_updated":"2024-07-22T10:56:19Z","snapshot_observed_at":"2026-08-16T13:58:45.757126Z","submitted_at":"2024-04-22T18:56:18Z","title":"M\\'elange: Cost Efficient Large Language Model Serving by Exploiting GPU Heterogeneity","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.14527","snapshot_observed_at":"2026-08-12T18:28:01.457638Z","title":"M\\’elange: Cost efficient large language model serving by exploiting gpu heterogeneity","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.457638Z"},"links":{"cited_paper":"/paper/2404.14527","citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:4b60d9e6571c4f9c04fef8c54e3f240f8c19706e3d698cda19dd4666762eee62","observation_id":"c2a87714-bfe2-4b13-bb56-e6dce973d287","resolution":{"observed_at":"2026-08-12T18:28:01.457638Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:02.256437Z","title":"Fastertransformer","venue":null,"work_id":"1e1e0ab3-6aa3-4f5f-b8e7-e0ba02175bdb","year":2021},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.462536Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:954331719d402bec221a07acf3af3944033fff12979bda9cc627785feb6d267e","observation_id":"b274c18f-8aa3-4049-8c3b-ef85184042f6","resolution":{"observed_at":"2026-08-12T18:28:02.261255Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:01.466924Z","title":"Efficient memory management for large language model serving with pagedattention","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.466924Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:e74808e2476544f2ae2e1d696792c4e9ee1eb2a5cede81f45a80e90d5c686d69","observation_id":"d72f1f46-7198-4ed8-b7e6-929dac152349","resolution":{"observed_at":"2026-08-12T18:28:01.466924Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:02.232619Z","title":null,"venue":null,"work_id":"87ea19c3-c5fd-45eb-8fd0-e8f5e02e0213","year":2023},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.471222Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:938b59b727fd7fadf4dd46eacafd4d36f0feb689bcf97dd2a0334aeef7830eb1","observation_id":"8059a009-b02e-4a5e-90ab-5b5462e55ffc","resolution":{"observed_at":"2026-08-12T18:28:02.236920Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:02.217200Z","title":"Huggingface text generation inference","venue":null,"work_id":"355f620a-13ed-48aa-93d9-b281891b87fa","year":2023},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.476060Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:0aab42c3f64a7afd79f99dd57d9c538a2d27e0d595f8d6d265bb113bf391cb43","observation_id":"55bbb4f4-33c7-4400-9a72-8dedbf17fc3d","resolution":{"observed_at":"2026-08-12T18:28:02.223059Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:02.200016Z","title":"Deepspeed inference","venue":null,"work_id":"aef0580e-fc43-4654-8c1a-3f44d3d6ef39","year":2022},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.481104Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:4588cd2e1d973a65888a9b48eb16b4bab1eaabe25114e101194a00e54492370b","observation_id":"8cf4ed1a-5c32-4f90-8fc4-6828229232e4","resolution":{"observed_at":"2026-08-12T18:28:02.206174Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:02.182548Z","title":"Tensorrt-llm","venue":null,"work_id":"60f37be1-fb12-4d1f-811a-b5432a0637c8","year":2023},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.486113Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:964d76a57fe9be158595f4c63e48a35e6caaf1a2169ab082afc87337a3556d1c","observation_id":"24929857-98da-480d-bb69-5d8d84af3978","resolution":{"observed_at":"2026-08-12T18:28:02.187894Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.15234","last_updated":"2025-07-23T10:11:55Z","snapshot_observed_at":"2026-08-16T14:32:00.300814Z","submitted_at":"2023-12-23T11:57:53Z","title":"Towards Efficient Generative Large Language Model Serving: A Survey from Algorithms to Systems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.15234","snapshot_observed_at":"2026-08-12T18:28:01.491752Z","title":"To- wards efficient generative large language model serving: A survey from algorithms to systems","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.491752Z"},"links":{"cited_paper":"/paper/2312.15234","citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:19cfa75bc8ec37909862c85625a5cd96e6d456b8807a3a12407b51e596120961","observation_id":"c33301c8-4278-4565-8476-81bae1b4adad","resolution":{"observed_at":"2026-08-12T18:28:01.491752Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:02.164151Z","title":"Kubernetes topology manager moves to beta","venue":null,"work_id":"3a4ceabd-4220-43e7-bf78-234f9dec4c61","year":2020},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.497305Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:7b9b6fe3f81ec61ff3c5ffb524c0e8d6745db54dd9ef3aa975f39d9f05d6eaa5","observation_id":"35e1fa7c-29a6-48ae-8bf0-d7501de9fdf6","resolution":{"observed_at":"2026-08-12T18:28:02.170281Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:02.145805Z","title":"Pod priority and preemption","venue":null,"work_id":"1b4055b5-fc6b-463f-b94b-308708a65e50","year":2024},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.502510Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:267cf597c9aa78e53c60ba9d97a404fb96af88f304dba0ebb3ed6d68bb38c5af","observation_id":"4a4273a1-70b4-4d0e-a9fb-d136a6eafbe2","resolution":{"observed_at":"2026-08-12T18:28:02.151560Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:02.126434Z","title":"Godel scheduler: a unified scheduler for online and offline tasks","venue":null,"work_id":"dedbccfa-3cbb-449b-a069-0c22f6124cd2","year":2023},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.507707Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:a4d74075c70ed0838eff8f1cb6d7f8c1f9683d8db463a087255c5fba25c2a0e2","observation_id":"56641ff5-55ad-4d69-a0ed-47f142ee90b4","resolution":{"observed_at":"2026-08-12T18:28:02.132252Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:02.105600Z","title":"Daemonset","venue":null,"work_id":"f0c20139-9cb4-43c3-a010-3937a6e56fe5","year":2024},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.513084Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:bcc57a4e138f12f0a8937349997a2e18e337672dcaa02e7a146eefb94b5637da","observation_id":"65743995-cd8e-4356-bc95-b0d38e2e1028","resolution":{"observed_at":"2026-08-12T18:28:02.112382Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:02.087421Z","title":"Kubernetes without kubelet","venue":null,"work_id":"10bfeabf-2cc7-49a9-bd0f-145bf12f5655","year":2024},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.518354Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:ae4be05f243ab85f97300c197976d6218ccee700991bb2f47e36cfad2d6e7793","observation_id":"38571f2a-df86-4258-95d3-9677b53151bd","resolution":{"observed_at":"2026-08-12T18:28:02.092846Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:02.070388Z","title":"Control topology management policies on a node","venue":null,"work_id":"0a598645-9138-4878-b6d6-ba3ad41b2544","year":2024},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.523308Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:bb0fd7592da0c4b0a1d1e54b0fc47ea81776c4508efa12022d2e0a1f1b706455","observation_id":"60eb0db1-4c97-4a32-ab61-2449af02ad9f","resolution":{"observed_at":"2026-08-12T18:28:02.076029Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:02.055812Z","title":"Towards {GPU} utilization prediction for cloud deep learning","venue":null,"work_id":"cdef53ed-d886-429b-806e-e6564252d53f","year":2020},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.528634Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:86338b055f10faa551f9a392becb2489cf9e392c3db39d69d9f64276526e071f","observation_id":"88a88c86-aed8-4e04-8cf0-847c994707ee","resolution":{"observed_at":"2026-08-12T18:28:02.060218Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:02.041730Z","title":"Horus: Interference-aware and prediction-based scheduling in deep learning systems","venue":null,"work_id":"016a723a-9e4d-49bb-9fc8-f9d99fb6afef","year":2021},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.533781Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:06170f35e639cd54c3c633a979c3dc47cd6db2f558e746a3bb83d3b3d25d13ff","observation_id":"b1a8bceb-c2a4-40f6-8530-0d45e9a5ee3f","resolution":{"observed_at":"2026-08-12T18:28:02.046202Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:02.026824Z","title":"Beware of fragmentation: Scheduling {GPU-Sharing} workloads with fragmentation gradient descent","venue":null,"work_id":"548bbfc5-1c5d-400f-9398-bb3034e4dec2","year":2023},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.539007Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:24522b6cfd9062d5cb9452594aed8757e8f83161c017598bc9dbf629a6b15713","observation_id":"925f7790-266f-4f84-b56e-a683690cf4c0","resolution":{"observed_at":"2026-08-12T18:28:02.031649Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:02.011835Z","title":"{HiveD}: Sharing a {GPU} cluster for deep learning with guarantees","venue":null,"work_id":"065295df-f0b1-47e0-b103-00325982bfbc","year":2020},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.543941Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:1ba865beea678fce43affa4d8c3d40dd3d4384d3ab9f8890497a64d6890e2f9a","observation_id":"b9e132b9-40af-44c1-bf05-587c08e6e83b","resolution":{"observed_at":"2026-08-12T18:28:02.016593Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:01.996920Z","title":"Supporting gpu sharing in cloud environments with a transparent runtime consolidation framework","venue":null,"work_id":"c348fa1c-630e-4ca0-a324-0ea4773ef0d4","year":2011},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.549509Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:8ede92d073c9e91028c9e7e7049d8517f043be6517d2723c20ab17ec1f17b8b1","observation_id":"a71a4c0e-220e-4f0c-b4f4-ff8a56479a6f","resolution":{"observed_at":"2026-08-12T18:28:02.001582Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:01.983115Z","title":"Fine-grained gpu sharing primitives for deep learning applications","venue":null,"work_id":"d780bbbd-1a74-41a4-ab62-a7827d506112","year":2020},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.554580Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:ebf7cdab8569ac81f8132788a8dbc51193722d7b94813989ed2228dba8d79c36","observation_id":"748ddaea-6be4-49c5-ac6c-c15e16774fd0","resolution":{"observed_at":"2026-08-12T18:28:01.987696Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:01.968746Z","title":"Gpushare: Fair-sharing middleware for gpu clouds","venue":null,"work_id":"dc38af48-b768-4747-a7ce-1c826771ff8f","year":2016},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.560039Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:9b3ad341b48218a421adbc2764b3fcb197730a6de658123ddb635cf7f9ed76f9","observation_id":"8223a453-87fd-43c1-93e8-110117107960","resolution":{"observed_at":"2026-08-12T18:28:01.973274Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:01.954845Z","title":"Nvidia cloud native technologies: Gpu sharing","venue":null,"work_id":"58d8181b-f9f3-4704-9c30-f22d55674bbe","year":2024},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.565070Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:b68f506d452187746e323b246909968100f3f15159e83d145a67cd9ecc979cd9","observation_id":"4506a771-c34a-4ba0-9258-99d951452b22","resolution":{"observed_at":"2026-08-12T18:28:01.959345Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:01.940926Z","title":"Advanced features in ibm power8 systems","venue":null,"work_id":"6e2a44be-9d42-4deb-9a7c-394f1d4b44b8","year":2015},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.570281Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:3182a399146a068f4158b90c8e178709c80627beadd5dce851f2c2273f17cd9a","observation_id":"4c09c855-110c-49b5-b9a1-3475a4e1f3c7","resolution":{"observed_at":"2026-08-12T18:28:01.945397Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:01.927620Z","title":"Performance evaluation of the nvidia tesla p100: Our directive-based partitioning and pipelining vs","venue":null,"work_id":"bbb1154f-8706-432f-bbb3-598353f47147","year":2017},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.575734Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:d7f8f77f0d0bb2ebbe90a6ab1640ccde1921beeb2a9f79e3a60e174693bfe3d8","observation_id":"1fb6f2b0-c3d3-45af-a511-191befec4d7f","resolution":{"observed_at":"2026-08-12T18:28:01.931958Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:01.912332Z","title":"Topology-aware scheduling framework for microservice applications in cloud","venue":null,"work_id":"eed685d7-1e4d-49ac-994b-779e5f59c10a","year":2023},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.580861Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:c8f4d831cee2af1794df215df2564152d7dd5cc794a57b06347279e715108351","observation_id":"deeb195f-5236-4900-8171-18ad8d71598f","resolution":{"observed_at":"2026-08-12T18:28:01.917790Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:01.896628Z","title":"Katalyst core","venue":null,"work_id":"61d15826-7eb3-461d-bc9b-952c66ff7f5c","year":2023},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.586803Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:bbf6ff5420164a9dea546810954ed0b5318653247b22fc2f9ed88d5175e292c9","observation_id":"dbf852d4-d6ba-44a7-a2ef-b31305b0d091","resolution":{"observed_at":"2026-08-12T18:28:01.901739Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:01.878438Z","title":"Topology-aware resource allocation for data-intensive workloads","venue":null,"work_id":"0851dae1-72f3-48b7-b65a-3edc22c6dc21","year":2010},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.592516Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:26520729e2375f6a4ee46f1370e714938c32a41128d10fc437087b79a603469c","observation_id":"93263d25-13c5-437c-9a69-5016325fcf88","resolution":{"observed_at":"2026-08-12T18:28:01.884085Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:01.860762Z","title":"Towards topology aware pre-emptive job scheduling with deep reinforcement learning","venue":null,"work_id":"bafaefcc-ee19-4e6c-aa28-2d182ab43e01","year":2020},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.597592Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:abaf45d72f16649112bf8986123feaf2204062cee3a2a012e10451a670e8fee0","observation_id":"6f4c2c72-2123-4979-afa7-c94df6b035d9","resolution":{"observed_at":"2026-08-12T18:28:01.865992Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:01.842844Z","title":"Microsecond-scale preemption for concurrent {GPU- accelerated}{DNN} inferences","venue":null,"work_id":"8e1fc4d2-9077-470c-985b-945409727cc2","year":2022},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.603019Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:d404b0156419064cf2083a687d839e20f727fec3bf2aacbd7fd3c570b2a93dbd","observation_id":"792f1e0a-0f86-4e97-a7ec-91a733a6150d","resolution":{"observed_at":"2026-08-12T18:28:01.849163Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:01.825208Z","title":"Efficiently programming large language models using sglang","venue":null,"work_id":"3f7c707a-0d1d-425d-b63f-27cae37a9bad","year":2023},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.608437Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:6287f9877e4fa3f4db9544f407d13d2dcaf7b400248b787f07fdc67a6cc89805","observation_id":"cc0d2b9d-a5d2-457a-9492-ffaf42566c25","resolution":{"observed_at":"2026-08-12T18:28:01.830860Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:01.614459Z","title":"Orca: A distributed serving system for {Transformer-Based} generative models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.614459Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:3e80bbc4de80ebb601e72221d4096e9bcff3265bff36ee8498b107afaf77f41c","observation_id":"d548ec38-5922-4dd3-826e-3af1d996ebbd","resolution":{"observed_at":"2026-08-12T18:28:01.614459Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05920","last_updated":"2024-09-25T05:57:51Z","snapshot_observed_at":"2026-08-16T21:53:57.298060Z","submitted_at":"2023-05-10T06:17:50Z","title":"Fast Distributed Inference Serving for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.05920","snapshot_observed_at":"2026-08-12T18:28:01.620434Z","title":"Fast distributed inference serving for large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.620434Z"},"links":{"cited_paper":"/paper/2305.05920","citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:03f8d8661f0acef5b66212adaadefb9649a4d4cda37b759d09aa999a722e9ccb","observation_id":"9ff1a8a3-098a-48de-9bd6-1c601afba6c4","resolution":{"observed_at":"2026-08-12T18:28:01.620434Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:01.796380Z","title":"Bert loses patience: Fast and robust inference with early exit","venue":null,"work_id":"3118af73-f094-46eb-8b91-cfbd3c83c536","year":2020},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.625813Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:682b0fa63c31b177af81cdd1ae366dcf988b6884c6d7f42f5d2458656846741d","observation_id":"799a9f9b-e866-443b-a8f8-1e266cf3c4c9","resolution":{"observed_at":"2026-08-12T18:28:01.801769Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:01.631009Z","title":"Flashattention: Fast and memory-efficient exact attention with io-awareness","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.631009Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:b8b452c9196c1926b5a392fa22e501060af32ecc259d6ed10f7246a5f88488fe","observation_id":"2ef5b5fb-8604-47f4-9f99-e25ed9042c7d","resolution":{"observed_at":"2026-08-12T18:28:01.631009Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:01.636569Z","title":"Sparsegpt: Massive language models can be accurately pruned in one-shot","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.636569Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:2038038eb7f16c47aa22bd7326bb6e4a3a7a18f79436d0f578541f06c2f557b8","observation_id":"a7a7f4e6-21d2-45c6-bc1d-78087adeda08","resolution":{"observed_at":"2026-08-12T18:28:01.636569Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:28:01.756077Z","title":"Awq: Activation-aware weight quantization for on-device llm compression and acceleration","venue":null,"work_id":"a5d191a3-4cf7-4cfc-90e2-a6e3d7b22a7f","year":2024},"citing_paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-12T18:28:01.641782Z"},"links":{"citing_paper":"/paper/2411.11560"},"observation_digest":"sha256:65dec574eedaf3c294bae5ca73b6fa7c14aac5b2f6547de3872464cb6223bc5c","observation_id":"563b8ecf-5b3a-47ba-af67-92051c370136","resolution":{"observed_at":"2026-08-12T18:28:01.763314Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2411.11560","last_updated":"2024-11-18T13:26:09Z","latest_version":1,"primary_category":"cs.DC","snapshot_observed_at":"2026-08-19T03:13:22.087421Z","submitted_at":"2024-11-18T13:26:09Z","title":"Topology-aware Preemptive Scheduling for Co-located LLM Workloads"},"reference_resolution":{"displayed":42,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":11,"verified_exact":0,"verified_fuzzy":31},"total_outbound_references":42},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 19 August 2026, this Paper Citation Record lists 42 of 42 outbound references and 0 inbound Pith citation observations for arXiv:2411.11560."}