{"as_of":"2026-08-09T22:49:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:5919d795165a07433090754240a74114aa66cd532d6def575a64ab98115c3c29","coverage":[{"denominator":64,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":64,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T14:45:20.821661Z","state":"measured"},{"denominator":64,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":64,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.18006/citation-record","integrity":"/paper/2507.18006/integrity","json":"/paper/2507.18006/citation-record.json","paper":"/paper/2507.18006"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-06T14:45:20.562559Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.562559Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:593c8528d14837dad8a4a8269c0c55f60f0715145e109da34a0f3d1af0aa845a","observation_id":"c2e070f6-fd83-452e-81fd-310b7c0dd7c0","resolution":{"observed_at":"2026-08-06T14:45:20.562559Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-06T14:45:20.567366Z","title":"Llama 2: Open foundation and fine- tuned chat models.https://arxiv.org/abs/2307.09288, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.567366Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:bdbd0e2b5fef7e34017483ec57fd1f16dfa1d3fcd4ba2ace2c704f56ca0d7c1f","observation_id":"f5dd5571-5204-4e44-b7d8-0eccd1211acf","resolution":{"observed_at":"2026-08-06T14:45:20.567366Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-06T14:45:20.571852Z","title":"Deepseek-v3 technical report.ArXiv, abs/2412.19437, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.571852Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:ce5abe07cac89223d6b29c6ed7fa2a76e9d9f47404c11bd1663de5699f39913b","observation_id":"603b8f94-dccd-40a7-8181-7e40f6adac5b","resolution":{"observed_at":"2026-08-06T14:45:20.571852Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2201.03514","last_updated":"2022-06-27T08:14:54Z","snapshot_observed_at":"2026-08-01T21:13:22.714773Z","submitted_at":"2022-01-10T18:17:05Z","title":"Black-Box Tuning for Language-Model-as-a-Service","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2201.03514","snapshot_observed_at":"2026-08-06T14:45:20.576394Z","title":"Black-box tuning for language-model-as-a-service.ArXiv, abs/2201.03514, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.576394Z"},"links":{"cited_paper":"/paper/2201.03514","citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:c00277e7a98b34e6f65ee46ca6bf6ba966a0a528d5c0bb3f51099eb7fea95ef8","observation_id":"738dbaeb-a711-43db-bfa1-62f073dc36c5","resolution":{"observed_at":"2026-08-06T14:45:20.576394Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:28.455276Z","title":"Xing, Hao Zhang, Joseph E","venue":null,"work_id":"bddc8d08-99fc-476c-adb6-9a82cbfc7efd","year":2023},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.581123Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:b4d8b12f978c02684099e844a7d847af684e8d10a56c9622120f1053eb5dc0bc","observation_id":"db8eb68e-5004-45fc-b8fa-c030f11af1f7","resolution":{"observed_at":"2026-08-06T14:45:28.556477Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.04226","last_updated":"2023-03-07T20:36:13Z","snapshot_observed_at":"2026-08-09T08:26:36.720121Z","submitted_at":"2023-03-07T20:36:13Z","title":"A Comprehensive Survey of AI-Generated Content (AIGC): A History of Generative AI from GAN to ChatGPT","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.04226","snapshot_observed_at":"2026-08-06T14:45:20.585569Z","title":"Yu, and Lichao Sun","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.585569Z"},"links":{"cited_paper":"/paper/2303.04226","citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:c0ab6eeeb13beadfb5039973a87ccdd214aaf97f2255fc86f75293a601e04e8a","observation_id":"55b91946-2e24-42ce-8a21-b0f42ff41d2e","resolution":{"observed_at":"2026-08-06T14:45:20.585569Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2107.03374","last_updated":"2021-07-14T17:16:02Z","snapshot_observed_at":"2026-08-08T11:58:24.516369Z","submitted_at":"2021-07-07T17:41:24Z","title":"Evaluating Large Language Models Trained on Code","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2107.03374","snapshot_observed_at":"2026-08-06T14:45:20.591384Z","title":"Evaluatinglarge language models trained on code.ArXiv, abs/2107.03374, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.591384Z"},"links":{"cited_paper":"/paper/2107.03374","citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:c467f6e4c6dd0f9e618a29cfbd8cc5108f54dd34e725815c5f444c147a168427","observation_id":"71922fdf-ee1e-4418-99a7-d8fa77b1303b","resolution":{"observed_at":"2026-08-06T14:45:20.591384Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12950","last_updated":"2024-01-31T19:47:26Z","snapshot_observed_at":"2026-07-06T16:10:07.931347Z","submitted_at":"2023-08-24T17:39:13Z","title":"Code Llama: Open Foundation Models for Code","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12950","snapshot_observed_at":"2026-08-06T14:45:20.595686Z","title":"Code llama: Open foundation models for code.ArXiv, abs/2308.12950, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.595686Z"},"links":{"cited_paper":"/paper/2308.12950","citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:187ddf20a4a431b3f96b7b8dda2040148ea8197cd31e6a2f0c14474372cb7a76","observation_id":"bdf5a873-497e-407e-862f-77e370601bbc","resolution":{"observed_at":"2026-08-06T14:45:20.595686Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:28.228332Z","title":"Accessed: Apr","venue":null,"work_id":"2c095341-7d4c-4ffb-96f7-a94edc02c3b5","year":2025},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.600056Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:290cb57ab06d71fa6d6912af6562b0032679c573bd9ee87fe95fda5ba911aa16","observation_id":"51fe50b3-27fa-4952-8078-3a7759ba61d3","resolution":{"observed_at":"2026-08-06T14:45:28.337931Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:28.084993Z","title":"Accessed: Apr","venue":null,"work_id":"e6dde3ea-5673-4789-a6d7-236fb2240ce6","year":2025},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.603833Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:ca720abd10c5e6d6387d369031c25cddf2f66e4778d5d02e576c4fc0314166c5","observation_id":"b7a24f66-cd92-485f-a4af-09779ae31ac7","resolution":{"observed_at":"2026-08-06T14:45:28.135926Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:20.607908Z","title":"Smoothquant: Accurate and efficient post-training quantization for large language models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.607908Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:e24180d7864f9fbd5a71edb28cee8d8e4445e671477089cdbbcd53c1bb933161","observation_id":"d5d67e8e-aa31-40b0-85f3-1718fcff4046","resolution":{"observed_at":"2026-08-06T14:45:20.607908Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:27.917140Z","title":"Plug-and-play: An efficient post-training pruningmethodforlargelanguagemodels","venue":null,"work_id":"4f3d616f-ae0d-47bf-b7fd-cae1680e2177","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.612417Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:6f0708ec3da27069e59c2933c2de240aae931f928d6cbe00e71993b2c6c7281c","observation_id":"8ed255ca-b436-478e-be84-cba993f9765f","resolution":{"observed_at":"2026-08-06T14:45:27.983499Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:27.740750Z","title":"Awq: Activation-aware weight quantization for llm compression and acceleration, 2024","venue":null,"work_id":"3b9ce18e-ed1e-46c2-a2f5-c014278813bc","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.616704Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:0dc25e61d745fadc385966032c82b67c909bf847a6a057364cedcd31e2c841ce","observation_id":"226107fc-4b94-4798-8697-ebada942d07b","resolution":{"observed_at":"2026-08-06T14:45:27.805979Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:27.604722Z","title":"Zico Kolter","venue":null,"work_id":"c34f751d-b6a9-4365-b29e-dc8f107a7d46","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.620795Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:9eb6196e1cac995ac691e576561b0f38ad0d2c81aa38f3052069346145cd2283","observation_id":"bbac1862-4e2a-4f20-9b8f-9b4d6315ab69","resolution":{"observed_at":"2026-08-06T14:45:27.665893Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:27.425450Z","title":"Multiplexing dynamic deep learning workloads with slo-awareness in gpu clusters","venue":null,"work_id":"0092d6e8-4269-4aef-a46f-a8ab30ffbd9e","year":2025},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.624755Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:7aaed7fe3b1958544a9863587a636e8c0f15f353695c784b9f1950abb61c75cb","observation_id":"7f6173b9-8bbc-4803-a0ec-d755b9c17f1b","resolution":{"observed_at":"2026-08-06T14:45:27.515390Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:27.236012Z","title":"Cloudnativesim: A toolkit for modeling and simulation of cloud-native applications","venue":null,"work_id":"11371e9b-b7e6-4fee-984a-fa21e3993b79","year":2025},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.628509Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:cad992eee837614d36ebbc844b744ab3753be6871764d601aff06dcd6080fef3","observation_id":"43514fe2-a55b-48e3-862c-9f96811117c8","resolution":{"observed_at":"2026-08-06T14:45:27.346301Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:27.069834Z","title":"Llminaflash:Efficientlargelanguagemodelinferencewith limited memory, 2024","venue":null,"work_id":"87bc7391-35fa-4cd6-acac-42f4ba2e5f02","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.632443Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:4527d2c23e41a9b7f3e5f371af2e88ab56b2d53fecafbe193b0157bb56cf227c","observation_id":"ab4ad36b-d4ea-4784-8c2b-96f02774f583","resolution":{"observed_at":"2026-08-06T14:45:27.162975Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:20.636501Z","title":"Spotserve: Serving generative large language models on preemptible instances","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.636501Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:881460bbbca04e74ee749a3620e1f6163909cbb6671451247c89890709d74f0e","observation_id":"59bf1481-178e-4fc5-b5d8-ae5b96ba2e8a","resolution":{"observed_at":"2026-08-06T14:45:20.636501Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:26.890465Z","title":"Llumnix:Dynamicschedulingforlargelanguage model serving","venue":null,"work_id":"d6bedaf8-ca6c-4c1e-bd2d-fab78dcdf541","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.640277Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:74b91647ebe23905d66f6cae788ec1a8c9b232aa1950b6cf1ff5f7cbe5058ac4","observation_id":"0736df2e-d30d-49e9-8f47-458134c404ed","resolution":{"observed_at":"2026-08-06T14:45:26.979547Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:26.701678Z","title":"Serving heterogeneous machine learning models on multi-gpu servers with spatio-temporal sharing","venue":null,"work_id":"9e53235e-9fa2-4d6c-adb9-967df0ac00f9","year":2022},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.643981Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:8a9a01ba860c869adb98d4d032ec2563af232e9d2ff77b5b522728bf148f3ef0","observation_id":"559ffcaf-15c7-4c90-8d29-7712367d1853","resolution":{"observed_at":"2026-08-06T14:45:26.787021Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:26.529937Z","title":"Inferline:latency-aware provisioningandscalingforpredictionservingpipelines.In Proceedings of the 11th ACM Symposium on Cloud Computing, pages 477–491, 2020","venue":null,"work_id":"f16b045a-d77e-4250-ad8d-3b63ef95ce4c","year":2020},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.647609Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:7bb63e7f6f3fa4cf75e4aad12b797090ef2581fcdc5d23e8033f99568ab9b940","observation_id":"9c86c373-b4bc-422f-b8be-c2e46e1f2c0b","resolution":{"observed_at":"2026-08-06T14:45:26.604327Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:26.350073Z","title":"Optimizing llm inference throughput via memory-aware and sla-constrained dynamic batching, 2025","venue":null,"work_id":"2f14bbc2-4922-41e2-aab3-dcc3c0f4fcfc","year":2025},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.651829Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:1bf5aa70d90bae0d8a9c6f1aece1a528040b44f660da35e51751169a0bc6b9b4","observation_id":"c8937796-4580-4d32-a58c-fee8349597d3","resolution":{"observed_at":"2026-08-06T14:45:26.431926Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:26.152376Z","title":"Alloystack: A library operating system for serverless workflow applications.Pro- ceedings of the Twentieth European Conference on Computer Systems, 2025","venue":null,"work_id":"0f48f668-16e2-4402-9c36-ec26cca048c8","year":2025},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.655748Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:4f61c7656d6a15e324632660f0fc62563b5783842c56857f6a637590ef6ae041","observation_id":"ae8538a9-5dcf-4c02-88d2-141892950ff8","resolution":{"observed_at":"2026-08-06T14:45:26.270323Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:25.932844Z","title":"Lora-flow: Dynamic lora fusion for large lan- guage models in generative tasks","venue":null,"work_id":"b5743c40-b20c-4f95-95a4-15ff1dd753dc","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.660020Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:8105414d22ed207f885c4b38b8f18d9106e483a593eb1f6945dbb1ce9e5a1b24","observation_id":"0777050f-f079-44e5-8789-ffd80508e5ea","resolution":{"observed_at":"2026-08-06T14:45:26.052390Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:25.742506Z","title":"Accessed: Apr","venue":null,"work_id":"125b83f0-d9e3-4912-894d-4634a1b78efb","year":2025},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.663827Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:8147db57423159ca6c196d4ae964c0c6d85976f87ba99e91b62f7360eab10284","observation_id":"f1344fe9-90df-4ba6-8d48-ab40f74d133f","resolution":{"observed_at":"2026-08-06T14:45:25.830080Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:25.567172Z","title":null,"venue":null,"work_id":"3f8fd8fe-189a-4a6b-8862-653fce8b6bfb","year":2020},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.667723Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:0ef994e93c4eedeb955c59ae7a1cba8d98e0a18b1b2990000fa5b9fa1ffd4945","observation_id":"01a69b05-5d9b-463a-b295-4217399f77da","resolution":{"observed_at":"2026-08-06T14:45:25.642539Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:25.404456Z","title":null,"venue":null,"work_id":"99ef0896-5a21-4f70-9276-3c30d6709540","year":2026},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.675885Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:d5bf09049891f1f03fe777ca3dd2fb2956299fbef2b35d106d6a0fb06e60d04b","observation_id":"2213fac0-1631-4eec-bdc7-3978d02e6221","resolution":{"observed_at":"2026-08-06T14:45:25.486001Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:20.681562Z","title":"Attention is all you need.Advances in neural information processing systems, 30, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.681562Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:e774f31e90c1f98949438abd7a8980b5a13784038449d2bab2fcca58ef1ce17b","observation_id":"1421a709-5b0d-4168-99c6-cff475463e4b","resolution":{"observed_at":"2026-08-06T14:45:20.681562Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:25.200271Z","title":"Llm inference serving: Survey of recent advances and opportunities, 2024","venue":null,"work_id":"9d24d35c-5822-429b-95ac-572e8df8f086","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.686206Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:be34c662eb11d780074eb60e34dd4a58defeacf709f3e9eb4e863ca13ee0883e","observation_id":"25fa635e-8597-4a2c-a108-282095a2f577","resolution":{"observed_at":"2026-08-06T14:45:25.265670Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:25.015072Z","title":"Optimizing mixture-of-experts inference time combining model deployment and communication scheduling, 2024","venue":null,"work_id":"f224f3ff-dee9-4fbe-b878-50bc30c65755","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.690334Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:d2136de30a07010988d734f9d5909f4743b3d22428cdcce4bc015cdf38804e1b","observation_id":"85001b70-bf8c-41bd-9932-9291d297961a","resolution":{"observed_at":"2026-08-06T14:45:25.101053Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:24.817562Z","title":"Mooncake: A kvcache-centric disag- gregated architecture for llm serving, 2024","venue":null,"work_id":"1448b175-25d0-4359-8fa8-54f5a296e421","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.694330Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:aa2483f0821f94fc5debd81dceba97d9f63440ba460c1c1c3524b4ecd6544229","observation_id":"3df35fa0-3e6a-4d50-95ce-866af36fc334","resolution":{"observed_at":"2026-08-06T14:45:24.909654Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:24.593539Z","title":"Spinfer: Leveraging low-level sparsity for efficient large language model inference on gpus","venue":null,"work_id":"b6c4de4c-8ef8-41e4-9796-28a287f9e3ed","year":2025},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.698281Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:b73c327fd1ab5ebe4850c1dfb721b1aa2859093e441947541cc9c2287e7212f9","observation_id":"ec14dcb2-cd43-4691-8eb5-08930a6dfee8","resolution":{"observed_at":"2026-08-06T14:45:24.699198Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:24.443357Z","title":"H2o:Heavy-hitteroracleforefficientgenerativeinference of large language models.Advances in Neural Information Processing Systems, 36:34661–34710, 2023","venue":null,"work_id":"d3bea8ba-ff2e-4623-9ecc-682b30e48b7e","year":2023},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.702463Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:988160ef358a3dc29bea53c9f24269f0fbb3733c8ffb47f8d9fb229d815b108d","observation_id":"1ecc3806-31e9-491d-ae80-e0550faa2b0b","resolution":{"observed_at":"2026-08-06T14:45:24.511967Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:24.188936Z","title":"Splitwise: Efficient gen- erative llm inference using phase splitting.2024 ACM/IEEE 51st Annual International Symposium on Computer Architecture (ISCA), pages 118–132, 2023","venue":null,"work_id":"82fd02ad-a25a-4b07-ac7d-b00208d14148","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.706607Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:784f53290b7d98f635593a95bf4bd72b7e11c728847bc070be02f3e3a780adff","observation_id":"76faaf30-a734-4153-8442-4a54223cfa77","resolution":{"observed_at":"2026-08-06T14:45:24.286745Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:23.993448Z","title":null,"venue":null,"work_id":"e40c27fe-6168-4d92-aa53-2bb49dd57db8","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.710774Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:8e8c488b36ab76f565d66ce7c956e584751fbcfc132bc255c6873e17190a58ee","observation_id":"c4c725c7-5a59-4d32-8bf4-3e95d7168ab7","resolution":{"observed_at":"2026-08-06T14:45:24.086241Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:23.751104Z","title":"Inference without interference: Disaggregate llm inference for mixed downstream workloads, 2024","venue":null,"work_id":"c1cdd3dd-8988-4066-92be-42f0fb76398f","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.714736Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:ad24d5c7d15d10f04620205f2f5b16ee392351458384af8583edd19e7917d095","observation_id":"43ce76af-0cca-4fae-bae4-1514640d0afa","resolution":{"observed_at":"2026-08-06T14:45:23.843614Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:23.542572Z","title":"Dynamollm:Designingllminferenceclustersforperformance and energy efficiency, 2024","venue":null,"work_id":"5ec41813-a7b0-4652-9b6f-59d788f941e7","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.719014Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:8fbc412d3b0c9a5321655f85c0477a96018ec3a09429412c3f1a9ed38fa6887c","observation_id":"904791a1-6169-4a39-a9f4-18cb5f46ec30","resolution":{"observed_at":"2026-08-06T14:45:23.669168Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:23.333940Z","title":"Skyserve:Servingaimodels across regions and clouds with spot instances","venue":null,"work_id":"02d67659-8402-4f50-a328-d9862b7de7f7","year":2025},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.722857Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:cc94dfdf2067572ef470250786e4cf91654a1fa927f2667ef4688c10e13740cf","observation_id":"d9ebb7f2-f091-404d-8285-19b0415de7b8","resolution":{"observed_at":"2026-08-06T14:45:23.437689Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:23.119873Z","title":"Towardsefficientandreliablellmserving: A real-world workload study.arXiv preprint arXiv:2402.XXXXX, 2024","venue":null,"work_id":"1ba45b8f-686a-46c8-990d-4695d481b355","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.727165Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:a9cc911f5abe2c4a141fb9789209459d50e9278c74146553953e97b5d2d34d2c","observation_id":"d650dde7-70e7-4800-9354-556790f414ee","resolution":{"observed_at":"2026-08-06T14:45:23.227733Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:22.917236Z","title":"Usher: Holistic interference avoidance for resource optimized ML inference","venue":null,"work_id":"2b2a710e-d63d-4e47-aeeb-d13acae71e55","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.731400Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:13e59b5be52bf503b5d7628ea730b0807ee31d72a36d5b59f37159fa7c696f84","observation_id":"50e2209c-e0bc-4164-972b-9d186e868a2e","resolution":{"observed_at":"2026-08-06T14:45:23.012903Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:22.712257Z","title":null,"venue":null,"work_id":"3a96d361-f111-4111-9521-b881ebfcb3cd","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.735534Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:8ee92e66a5a5b1bdf9f79f12ec533008d2757ca0a22f83f9547b7413bd610f3f","observation_id":"71fe8728-f510-4dde-8c3a-93b4cea88868","resolution":{"observed_at":"2026-08-06T14:45:22.779244Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:22.472864Z","title":"Distserve: Disaggregating prefill and decoding for goodput-optimized large language model serving","venue":null,"work_id":"ff9513b3-0ee1-4649-9c86-eb3e44db5cd2","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.739514Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:b5a1d1bcfbf65b9276957cb380857787522e38400d27b3dc4d2296f138b44589","observation_id":"515dbeab-5865-4a1e-89d1-4b2dbc2a8595","resolution":{"observed_at":"2026-08-06T14:45:22.599134Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:22.314614Z","title":"Alpaserve: Statistical multiplexing with model parallelism for deep learning serving","venue":null,"work_id":"d1f61fe4-abef-49ed-88fe-fcb2a5fa6ca7","year":2023},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.743488Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:a0d58091d5a60b22e1ceab5300a1051051971d256a49fbc471eb4b818a7a417b","observation_id":"b17279ee-733a-4137-87dc-7c2105163f89","resolution":{"observed_at":"2026-08-06T14:45:22.387183Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:22.108310Z","title":null,"venue":null,"work_id":"fd5106b5-2045-4cc8-8d31-b2728d087b2b","year":2020},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.747191Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:49cc6dc2d88d8da53172e8aa5b447831da73f49b6783a6d2bb3f9355a02b8a82","observation_id":"3db551ec-4c4a-449c-8a13-15ea022f6fdf","resolution":{"observed_at":"2026-08-06T14:45:22.205088Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.935781Z","title":"Yadwadkar, and Christos Kozyrakis","venue":null,"work_id":"9e276474-2fe1-4202-a25f-7be9b4f05658","year":null},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.750754Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:a87316233326a33bd7d0a633a787db74044603eb95e85f7f33a06aa66bb6b16e","observation_id":"8e952f21-82e8-4dfb-aeab-26bb3c1c3e81","resolution":{"observed_at":"2026-08-06T14:45:21.999977Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.05198","last_updated":"2022-05-10T22:40:17Z","snapshot_observed_at":"2026-07-06T13:08:41.837840Z","submitted_at":"2022-05-10T22:40:17Z","title":"Reducing Activation Recomputation in Large Transformer Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.05198","snapshot_observed_at":"2026-08-06T14:45:20.758677Z","title":"McAfee,MichaelAndersch,MohammadShoeybi,andBryanCatanzaro","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.758677Z"},"links":{"cited_paper":"/paper/2205.05198","citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:81f21d34907e3262dea774c4be7b7d45526b5ad35dbb45ef8baffa2d172e70a5","observation_id":"29feca7a-4ba7-4d1c-870a-f553e4fdb36f","resolution":{"observed_at":"2026-08-06T14:45:20.758677Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.535471Z","title":"On parallel processing systems: Amdahl’s law generalized and some results on optimal design","venue":null,"work_id":"0d751978-721a-42d5-a4b5-f9bc21adf0fa","year":1992},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.762745Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:d6e4e74453c2e79a03bb7d14ea6500dfbc2ee6d5a7d840c0422993fa1956b36f","observation_id":"6192e0ab-e576-407f-938f-4a21570f9f37","resolution":{"observed_at":"2026-08-06T14:45:21.610721Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.330409Z","title":"xformers: A modular and hackable transformer modelling library.https://github.com/facebookresearch/ xformers, 2022","venue":null,"work_id":"dabfc945-d31b-45a6-9169-5553adf6e5d9","year":2022},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.766581Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:fe73b9fe6bd98291df4c174afbfd8d38244fe6118eccbabad14805f9ee32087d","observation_id":"3607326f-b68b-41b0-bfc0-29a9cb5085ec","resolution":{"observed_at":"2026-08-06T14:45:21.412262Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.305012Z","title":"https://developer.nvidia.com/management-library-nvml, 2025","venue":null,"work_id":"4e755e41-d952-4ebf-916c-8f76315acea8","year":2025},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.770284Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:fbfd2da7a5a4c6861cb33fc05fb4202bf773f7ae6520e2147df15ae9c17390ba","observation_id":"b995e9ff-346e-43c5-a422-8101a97dc13e","resolution":{"observed_at":"2026-08-06T14:45:21.313583Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.272261Z","title":"Orca: A distributed serving system for transformer- based generative models","venue":null,"work_id":"55ac6ba0-20e8-4e53-ba66-36957a4d1409","year":2022},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.774358Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:b71ee8f2181099b07af60f0b9d5d193bc9f4661050a7f7d2711268c801b3b30d","observation_id":"33dca7d2-896f-4afa-a787-f6a83496e92f","resolution":{"observed_at":"2026-08-06T14:45:21.283738Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.231279Z","title":"Uellm: A unified and efficient approach for large language model inference serving","venue":null,"work_id":"720a43b6-dea3-4919-a8e2-0316374c801b","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.778185Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:7e1eb2a3a50637898f5de2831b6fe054e1b4136414aedef54140df5b0deab3c3","observation_id":"080e1029-b4cb-43b7-9733-66fb8c5041d6","resolution":{"observed_at":"2026-08-06T14:45:21.245867Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.185558Z","title":"Stanford alpaca: An instruction- following llama model.https://github.com/tatsu-lab/stanford_alpaca,","venue":null,"work_id":"24e2a8eb-a32e-43b6-823f-08a9037cb252","year":null},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.781963Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:386e08333a02d10db26bce6921643f34a0e6baebb775048995340ba8b594d05e","observation_id":"85ae5a38-9a43-400b-9e93-4dcf3748b8cf","resolution":{"observed_at":"2026-08-06T14:45:21.202598Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.096825Z","title":"Mepipe: Democratizing llm training with memory- efficient slice-level pipeline scheduling on cost-effective accelerators","venue":null,"work_id":"9df4ad4a-b9db-46dc-8b95-03a52ff6145e","year":2025},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.789659Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:b176a7bc16f9f891fe4be715a0cdeeae8fa0c240bed6b3e8f3d46cf25fbac01a","observation_id":"f8ce4997-8690-4e79-92bb-6d436cbc5305","resolution":{"observed_at":"2026-08-06T14:45:21.113047Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.065432Z","title":"Le, and Z","venue":null,"work_id":"6b1b471b-39b0-444d-ad54-f032e29adca6","year":2018},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.793446Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:1c32a7374db0771e98d51599ed50d845fab4933546b7f9cb254c2d3f31678384","observation_id":"d44c4132-fcb8-4e9b-b5eb-fe849f84db51","resolution":{"observed_at":"2026-08-06T14:45:21.078885Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.051154Z","title":"Mist:Efficientdistributedtrainingoflargelanguagemodelsviamemory- parallelism co-optimization","venue":null,"work_id":"21746a6f-4254-4c80-9ec1-6b425d5e67a6","year":2025},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.797367Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:6618dc544733b98d06f31dc395524085b519da0a86960182e413adc97ae22f52","observation_id":"ab48421c-b07c-4781-905c-e160553b1ac6","resolution":{"observed_at":"2026-08-06T14:45:21.055530Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.08053","last_updated":"2020-03-13T23:45:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-09-17T19:42:54Z","title":"Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.08053","snapshot_observed_at":"2026-08-06T14:45:20.801131Z","title":"14 Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling EuroSys ’26, April 13–April 16, 2026, Edinburgh, UK","venue":null,"work_id":null,"year":1909},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.801131Z"},"links":{"cited_paper":"/paper/1909.08053","citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:b2fa74c0bae05b93f63bef9a3d71b727f6568eed00082bad93a055e6ebbafa63","observation_id":"b2b93895-5df9-4f87-91b4-4b5a78aab060","resolution":{"observed_at":"2026-08-06T14:45:20.801131Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.037879Z","title":"Alpa: Automating inter and intra- operator parallelism for distributed deep learning","venue":null,"work_id":"e7c17256-3a88-4668-85cc-69e39493f87f","year":2022},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.805316Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:2166bf7ecc5971f3beb314a88eb3809b1365f72b8a20f4acdd2e1f268d7ca550","observation_id":"848a515f-29f3-418e-8124-111d46d06335","resolution":{"observed_at":"2026-08-06T14:45:21.041953Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.023917Z","title":"Fast state restoration in llm serving with hcache","venue":null,"work_id":"8cfe9f7b-43d8-43c9-bb63-7ce5de673b0b","year":2025},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.809631Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:017269c2376bf6db3da60b89a12277cea563464e10d7414526152a6c3ff0e76a","observation_id":"a7567b5a-d5ea-4aff-aee2-d1c5c27566fc","resolution":{"observed_at":"2026-08-06T14:45:21.028193Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.009772Z","title":"Fast and live model auto scaling with o(1) host caching, 2024","venue":null,"work_id":"a3b3460f-f2a5-4d0f-874c-c85138451556","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.813723Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:f22998eb0f7c929e0e228bd08edeab3beb709c3a19f4b04ec3a746706c78643e","observation_id":"5b2ca156-3b79-4afc-8b43-a3107e2d25aa","resolution":{"observed_at":"2026-08-06T14:45:21.013948Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:20.817827Z","title":"Deepspeed: System optimizations enable training deep learning models with over 100 billion parameters","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.817827Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:8cf0ae8f099f179319f12ce24fc411cd5ba96177b4066a1321e8bd5a75e809e9","observation_id":"106593b2-91f9-4854-87af-93f5faf2cc46","resolution":{"observed_at":"2026-08-06T14:45:20.817827Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:20.985086Z","title":"Infinigen: Efficientgenerativeinferenceoflargelanguagemodelswithdynamickv cache management","venue":null,"work_id":"68e06269-d815-474a-bc3a-cb0390e5bfad","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.821661Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:a2c7f806e69c8a3bb46aa1aa4f914806ed3669f913111875a87006f1d9d6aef7","observation_id":"d07250cf-5f05-4c41-8fdb-e4ecfcde3f15","resolution":{"observed_at":"2026-08-06T14:45:20.991136Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.722126Z","title":null,"venue":null,"work_id":"3eb55733-6552-4e63-93f8-414c416079d0","year":2021},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":411,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.754810Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:73dfb5ad4d219c92ace180f52cfdfdd7e0463ec1707acc234c4f6b1f7b8b77c5","observation_id":"bb71e73a-2187-431a-8428-e8c3931b8909","resolution":{"observed_at":"2026-08-06T14:45:21.824444Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:20.671778Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.671778Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:f00994656b653039176b54b10ad657ad2c838a14e2192a50d37d008c60471af4","observation_id":"a18aea7a-6abb-454a-a786-988f05df980b","resolution":{"observed_at":"2026-08-06T14:45:20.671778Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.127827Z","title":null,"venue":null,"work_id":"10b6ee2c-e93b-4b54-aa96-5e824aff3f72","year":2025},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.785956Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:2198c862c522788753dd336d1c66328ff0c6ddd35977651ba92cf43b1f50d1fb","observation_id":"2ef497c9-742a-44aa-9990-dbfd9f1c122c","resolution":{"observed_at":"2026-08-06T14:45:21.165178Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","latest_version":1,"primary_category":"cs.DC","snapshot_observed_at":"2026-08-09T08:27:15.038300Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling"},"reference_resolution":{"displayed":64,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":21,"verified_exact":0,"verified_fuzzy":43},"total_outbound_references":64},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 64 of 64 outbound references and 0 inbound Pith citation observations for arXiv:2507.18006."}