{"as_of":"2026-08-09T11:01:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:1a40ef40ff888f4c58dd406f93c741812715b938f5693b0b58d279157f05d8f6","coverage":[{"denominator":39,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":39,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-03T04:22:54.961186Z","state":"measured"},{"denominator":39,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":39,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.29575/citation-record","integrity":"/paper/2607.29575/integrity","json":"/paper/2607.29575/citation-record.json","paper":"/paper/2607.29575"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:51.621032Z","title":"The rapid adoption of generative ai,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:51.621032Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:d2ecadeeffc7d976acfd6b9a78885b12dd04cd2df6249d932b13aadd1136a9f8","observation_id":"1283f76c-0bbf-4207-80e6-721c88cf4e48","resolution":{"observed_at":"2026-08-03T04:22:51.621032Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:51.687853Z","title":"The adoption of chatgpt,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:51.687853Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:600d2ec8c2f903ef9bace23b795851bdcdd24c7fe6eaf081210caf625785354f","observation_id":"71750f8e-e65d-4dfc-8fc7-5c8e831faba4","resolution":{"observed_at":"2026-08-03T04:22:51.687853Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:51.737457Z","title":"Quantifying large language model usage in scientific papers,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:51.737457Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:ea9ce2a94538c1fc7f36d12085103cb3f10dad7ec6098f38757ef6e7356ce851","observation_id":"5e3aec83-1a24-4388-adde-a9bb58b1fdf0","resolution":{"observed_at":"2026-08-03T04:22:51.737457Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:51.844613Z","title":"Llumnix: Dynamic scheduling for large language model serving,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:51.844613Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:8b7942910d87f29049f97376c2f4a9627137b4c909525fc50c6d86a76f346951","observation_id":"d7a31b72-eb9d-4ed6-8138-bb0287ef2850","resolution":{"observed_at":"2026-08-03T04:22:51.844613Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:51.935319Z","title":"Sageserve: Optimizing llm serving on cloud data centers with forecast aware auto-scaling,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:51.935319Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:ebb0de39f1cdcd6f55c8c929e77554370252bb0f938799550d47d3666fdebcda","observation_id":"a627426c-a4f4-4206-a394-04d805de12ae","resolution":{"observed_at":"2026-08-03T04:22:51.935319Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.04466","last_updated":"2025-06-13T12:20:54Z","snapshot_observed_at":"2026-08-08T03:21:26.353063Z","submitted_at":"2024-10-06T12:42:04Z","title":"Large Language Model Inference Acceleration: A Comprehensive Hardware Perspective","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.04466","snapshot_observed_at":"2026-08-03T04:22:52.032744Z","title":"Large language model inference acceleration: A comprehensive hardware perspective,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:52.032744Z"},"links":{"cited_paper":"/paper/2410.04466","citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:5bce84b6436affb26ce508acbd7facc2f47bfbe8e4ec34babfd95c64e2356398","observation_id":"dc0fee93-9e4d-46c1-8b28-cfa5f828b8ae","resolution":{"observed_at":"2026-08-03T04:22:52.032744Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:52.082197Z","title":"Efficiently scaling transformer inference,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:52.082197Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:bb317a47ddc09139685ca1f7018b76f0f320fd856f0a5120f822e75eb09d9830","observation_id":"3d7e53ee-93d8-47cb-a734-76e6e4e5c880","resolution":{"observed_at":"2026-08-03T04:22:52.082197Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:52.167267Z","title":"Efficient llm inference: Bandwidth, compute, synchronization, and capacity are all you need,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:52.167267Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:00d02c97b3e64d60170302b1f0c71382696510c8aaf36c9e4089da69e38581e6","observation_id":"dbd5fb7e-2239-4859-89da-77361ba9beb1","resolution":{"observed_at":"2026-08-03T04:22:52.167267Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:52.228945Z","title":"Mind the memory gap: Unveiling gpu bot- tlenecks in large-batch llm inference,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:52.228945Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:52125893ca279ec27837edcae15e09232806d0bd6220d120d8d2602e70566721","observation_id":"8401ac48-13a1-42e6-96f7-d75a09241a32","resolution":{"observed_at":"2026-08-03T04:22:52.228945Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:52.312448Z","title":"Llmvisor: A real-time latency attribution model for multi-tenant llm serving,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:52.312448Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:1c2ebd55ea7f5f32432d24fea14af3147d4d8b2a253fe60bc4088a22e60464a4","observation_id":"e9be6525-2ed8-4deb-baa1-20852599966e","resolution":{"observed_at":"2026-08-03T04:22:52.312448Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:52.386461Z","title":"Predicting llm inference latency: A roofline-driven ml method,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:52.386461Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:3155df0272f75a642e0fcaa0ca63a05176752ad67a4c9a5b1cdb1d78a81e9e61","observation_id":"169dd382-9cd3-48bf-8c43-d32d50983289","resolution":{"observed_at":"2026-08-03T04:22:52.386461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:52.529800Z","title":"Language mod- els are few-shot learners,","venue":null,"work_id":null,"year":1901},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:52.529800Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:eac3469d80846e230d33a24652d251dac9fa3bf0c080d9ff88e76926f0eab719","observation_id":"faa6fa12-8937-465d-87bf-2c10a68eb48a","resolution":{"observed_at":"2026-08-03T04:22:52.529800Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-03T04:22:52.616647Z","title":"Llama: Open and efficient foundation language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:52.616647Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:7097680db3fe1d173ba14a3e34c97626ee74849a485e4212d02a4899b3917c3e","observation_id":"06875560-041e-47bd-bca5-4e0a529eaf44","resolution":{"observed_at":"2026-08-03T04:22:52.616647Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:52.703801Z","title":"Attention is all you need,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:52.703801Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:fb2baf6a8cd99a6cf297d00969ad3f81299b53be80122fe06e021d79abf7c997","observation_id":"ab551368-fb6b-46ed-b8d7-789c77352651","resolution":{"observed_at":"2026-08-03T04:22:52.703801Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:52.769107Z","title":"Orca: A distributed serving system for transformer-based generative models,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:52.769107Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:51960f0acf480c3f9ffb6f3f3d0874e9822646d4f2366bf677bf9e1bc12830a5","observation_id":"de2897af-d4a1-4f69-8e86-b899daadc481","resolution":{"observed_at":"2026-08-03T04:22:52.769107Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:52.843887Z","title":"Efficient memory management for large language model serving with pagedattention,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:52.843887Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:30c3c78f0895aae61dabac992bdaff0a5ea244d6299affccc40fbbdffbd11172","observation_id":"56f5eeae-c9e5-433a-b6c4-22b4f42723c9","resolution":{"observed_at":"2026-08-03T04:22:52.843887Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:52.928323Z","title":"TensorRT-LLM,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:52.928323Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:dc16a2d4be360194ff1746d02676ede843d08e30ade27ac6254687a83b6acc21","observation_id":"3aa6c78b-0554-4beb-a73e-df1f537fc2e2","resolution":{"observed_at":"2026-08-03T04:22:52.928323Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:53.030125Z","title":"DeepSpeed-MII,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:53.030125Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:f01015865a2c2c3bb59258eadffd960e684f877c2948bab24636d701717248ec","observation_id":"94854dc3-b780-4314-b2a6-ef0ecc871228","resolution":{"observed_at":"2026-08-03T04:22:53.030125Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:53.117570Z","title":"Slora: Scalable serving of thousands of lora adapters,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:53.117570Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:01f0cc629bea25d08fcd4579cc3d9f603eeab81752d3d47dd69116de4008c578","observation_id":"f142fcee-c690-41b7-a722-d974ca81131e","resolution":{"observed_at":"2026-08-03T04:22:53.117570Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.16363","last_updated":"2024-05-01T20:42:28Z","snapshot_observed_at":"2026-08-01T22:52:35.092898Z","submitted_at":"2024-02-26T07:33:05Z","title":"LLM Inference Unveiled: Survey and Roofline Model Insights","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.16363","snapshot_observed_at":"2026-08-03T04:22:53.202838Z","title":"Llm inference unveiled: Survey and roofline model insights,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:53.202838Z"},"links":{"cited_paper":"/paper/2402.16363","citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:b443929bdc9b94d973d2db95ac1e8dbf4f9c12b5512aa62ce28b235650fe844b","observation_id":"a133c8c9-ab5f-42b6-a05b-570a8d2e5d44","resolution":{"observed_at":"2026-08-03T04:22:53.202838Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:53.305385Z","title":"Taming{Throughput-Latency}tradeoff in{LLM}inference with{Sarathi-Serve},","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:53.305385Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:198f22597997d0f55d9a1e86f1c3ae09b9dff33753f7addec979ada387b2485e","observation_id":"46ded418-0c69-4282-b2fd-5ca8d11f8edc","resolution":{"observed_at":"2026-08-03T04:22:53.305385Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:53.428587Z","title":"Fast inference from transform- ers via speculative decoding,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:53.428587Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:e041010c1e785bf51d973008745f0fa25ed9d6fa4c778dbd5f334a5e673244df","observation_id":"e556d816-ff08-40d6-835e-f20286af23c8","resolution":{"observed_at":"2026-08-03T04:22:53.428587Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.01318","last_updated":"2023-02-02T18:44:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-02T18:44:11Z","title":"Accelerating Large Language Model Decoding with Speculative Sampling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.01318","snapshot_observed_at":"2026-08-03T04:22:53.550502Z","title":"Accelerating large language model decoding with speculative sam- pling,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:53.550502Z"},"links":{"cited_paper":"/paper/2302.01318","citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:f14e9d257d30d0462da7d507763e9d78f6ad8296407a6ef1c251b58abff96f3d","observation_id":"11947e0b-a77a-40a4-9b7f-51eeb4fa4fc2","resolution":{"observed_at":"2026-08-03T04:22:53.550502Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1911.02150","last_updated":"2019-11-06T00:19:05Z","snapshot_observed_at":"2026-07-06T08:35:01.386074Z","submitted_at":"2019-11-06T00:19:05Z","title":"Fast Transformer Decoding: One Write-Head is All You Need","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1911.02150","snapshot_observed_at":"2026-08-03T04:22:53.668526Z","title":"Fast transformer decoding: One write-head is all you need,","venue":null,"work_id":null,"year":1911},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:53.668526Z"},"links":{"cited_paper":"/paper/1911.02150","citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:3ef13a3f2c730e662cd588dd1a2a8026bd2f53235ae69be4704d64c09cebf45a","observation_id":"f0c8d29b-ce4a-4423-b2d6-fc9afdaae304","resolution":{"observed_at":"2026-08-03T04:22:53.668526Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.13245","last_updated":"2023-12-23T17:55:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-22T17:16:38Z","title":"GQA: Training Generalized Multi-Query Transformer Models from Multi-Head Checkpoints","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.13245","snapshot_observed_at":"2026-08-03T04:22:53.767380Z","title":"Gqa: Training generalized multi-query transformer models from multi-head checkpoints,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:53.767380Z"},"links":{"cited_paper":"/paper/2305.13245","citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:57a123639c5644f446c0eb45cbba17e78016ac331e766a1539eea2f1960ecf3d","observation_id":"4fff4c06-15c0-4d6c-82a7-e8a0c5e61c97","resolution":{"observed_at":"2026-08-03T04:22:53.767380Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:53.847735Z","title":"Awq: Activation-aware weight quanti- zation for on-device llm compression and acceleration,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:53.847735Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:6ffcb83ea279c1d4723c860f111302aa91864e27ccc59de5f9f188913809771a","observation_id":"8a9d34a5-5e25-41b3-b4f1-f9dde91eb23f","resolution":{"observed_at":"2026-08-03T04:22:53.847735Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.17323","last_updated":"2023-03-22T13:10:47Z","snapshot_observed_at":"2026-08-07T08:38:54.025062Z","submitted_at":"2022-10-31T13:42:40Z","title":"GPTQ: Accurate Post-Training Quantization for Generative Pre-trained Transformers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.17323","snapshot_observed_at":"2026-08-03T04:22:53.915456Z","title":"Gptq: Accurate post-training quantization for generative pre-trained transformers,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:53.915456Z"},"links":{"cited_paper":"/paper/2210.17323","citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:ebc5e83641dac7eb1f732856b61049c13e9b1737ecaa52a2f205c2927f5e7ee1","observation_id":"6f35459a-2052-4824-bc59-7fc008a098fc","resolution":{"observed_at":"2026-08-03T04:22:53.915456Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:54.024548Z","title":"Vidur: A large-scale simulation framework for llm inference,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:54.024548Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:a2b9b40d33da2c422f9ce1ad209ad218a1e7825ecd9090868b0a891ffc856d1b","observation_id":"ec390e4f-1a26-45ca-b8ef-c96e3928b659","resolution":{"observed_at":"2026-08-03T04:22:54.024548Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01698","last_updated":"2025-05-15T02:46:53Z","snapshot_observed_at":"2026-08-08T08:03:05.182888Z","submitted_at":"2024-06-03T18:00:50Z","title":"Demystifying AI Platform Design for Distributed Inference of Next-Generation LLM models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01698","snapshot_observed_at":"2026-08-03T04:22:54.136193Z","title":"Demystifying ai platform design for distributed inference of next-generation llm models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:54.136193Z"},"links":{"cited_paper":"/paper/2406.01698","citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:628d1cdf2d5b682f8cdbeb6844c668c63ed968c3587b207d8a90c821bec72e83","observation_id":"9474e2ac-8847-448d-a2c5-33784756ba3a","resolution":{"observed_at":"2026-08-03T04:22:54.136193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:54.209163Z","title":"Llmcompass: Enabling efficient hardware design for large language model inference,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:54.209163Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:39dae6503134ef18d78cef00e205099f414cc5a86d9f5f6270e6d7fa28f42ccf","observation_id":"e3a09dfd-6bd0-4c5c-b9c0-dad1f0f4702a","resolution":{"observed_at":"2026-08-03T04:22:54.209163Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:54.321060Z","title":"Amali: An analytical model for accurately modeling llm inference on modern gpus,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:54.321060Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:edf2a71907577c827c9264eb13ad4756cf6566b30f274e8a6eafd58ceb102147","observation_id":"8d956b91-494c-4617-83d1-4b357831ad48","resolution":{"observed_at":"2026-08-03T04:22:54.321060Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:54.393342Z","title":"Fairness in serving large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:54.393342Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:f5ea1d137d25ea473e33cfc434689636cd4347ba6e2bc358c05f7c79f45eae04","observation_id":"46db8890-3e1f-4963-9f18-6cc14c0f8fef","resolution":{"observed_at":"2026-08-03T04:22:54.393342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:54.458412Z","title":"Clean sharegpt dataset,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:54.458412Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:88ffd1a55446a0924119af561dc9ea1d936f68fe0865f45d4e5db5dfda3b6fc3","observation_id":"ef1c6471-f486-4f1d-a3b2-dc42413c0c53","resolution":{"observed_at":"2026-08-03T04:22:54.458412Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06825","last_updated":"2023-10-10T17:54:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T17:54:58Z","title":"Mistral 7B","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06825","snapshot_observed_at":"2026-08-03T04:22:54.533081Z","title":"Mistral 7b,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:54.533081Z"},"links":{"cited_paper":"/paper/2310.06825","citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:986e8f46113b24138e1f4ee6f941c7ab250c751c2deea40d446d5f2cbf06b50f","observation_id":"6b314849-88b5-42a6-910a-381b7b4b23f0","resolution":{"observed_at":"2026-08-03T04:22:54.533081Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.04324","last_updated":"2024-05-07T13:50:40Z","snapshot_observed_at":"2026-07-06T18:11:01.734775Z","submitted_at":"2024-05-07T13:50:40Z","title":"Granite Code Models: A Family of Open Foundation Models for Code Intelligence","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.04324","snapshot_observed_at":"2026-08-03T04:22:54.615902Z","title":"Granite code models: A family of open foundation models for code intelligence,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:54.615902Z"},"links":{"cited_paper":"/paper/2405.04324","citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:b6843c7d1fbe380162ab4688d7c227ff4e01880add3dbc5de6d2988cf74e2606","observation_id":"a2a6e250-6339-4122-b15f-b4bac850f715","resolution":{"observed_at":"2026-08-03T04:22:54.615902Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:54.707629Z","title":"Opt: Open pre-trained transformer language models,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:54.707629Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:4e7cc5d4593d5706ce0dd89a31d26a876c1123d73c58e3dfa27301bbdd1c44ff","observation_id":"ab58eaea-b3aa-4790-bff5-1bc003501ea7","resolution":{"observed_at":"2026-08-03T04:22:54.707629Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10671","last_updated":"2024-09-10T13:25:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-15T12:35:42Z","title":"Qwen2 Technical Report","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10671","snapshot_observed_at":"2026-08-03T04:22:54.849854Z","title":"Qwen2 technical report,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:54.849854Z"},"links":{"cited_paper":"/paper/2407.10671","citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:2eacb26f52193061d91e184e5206b05ef2ddf1c287d9876fc17542d9d6d3b320","observation_id":"543d0be1-ba38-436c-9a08-fd300a83dc8a","resolution":{"observed_at":"2026-08-03T04:22:54.849854Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:54.961186Z","title":"Ai and memory wall,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:54.961186Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:b37dfd2a1a230bfbd1bc0ef6c9d81c567f7d8c86304c22d144504aff45040279","observation_id":"c589949a-a181-4c18-b42c-a44f50ab1084","resolution":{"observed_at":"2026-08-03T04:22:54.961186Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.01068","last_updated":"2022-06-21T17:04:40Z","snapshot_observed_at":"2026-08-06T03:13:37.403059Z","submitted_at":"2022-05-02T17:49:50Z","title":"OPT: Open Pre-trained Transformer Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.01068","snapshot_observed_at":"2026-08-03T04:22:54.775813Z","title":"Available: https://arxiv.org/abs/2205.01068","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:54.775813Z"},"links":{"cited_paper":"/paper/2205.01068","citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:4aebf121c75d6537e6e0610822db8c17419242c2ae38bfb1f1838edf03bcb6f4","observation_id":"faa8d4de-c4d2-4dc7-919c-4814191e17ec","resolution":{"observed_at":"2026-08-03T04:22:54.775813Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","latest_version":1,"primary_category":"cs.DC","snapshot_observed_at":"2026-08-08T23:05:50.141195Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving"},"reference_resolution":{"displayed":39,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":39,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":39},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 39 of 39 outbound references and 0 inbound Pith citation observations for arXiv:2607.29575."}