{"as_of":"2026-08-11T09:48:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:2d6b20d04b3fde6faf88303757775a78b2aaf554d5faa9ac5c6d530833a926d1","coverage":[{"denominator":48,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":48,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T05:05:22.596292Z","state":"measured"},{"denominator":49,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":49,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-28T16:37:20.774251Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-01T21:36:15.531550Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"cited_work":{"arxiv_id":"2412.18106","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.18106","snapshot_observed_at":"2026-07-01T21:36:15.531550Z","title":"Tackling the dynamicity in a produc- tion LLM serving system with SOTA optimizations via hybrid pre- fill/decode/verify scheduling on efficient Meta-kernels","venue":null,"work_id":"9fdf95cb-2768-4030-92ae-a4899f51930e","year":2024},"citing_paper":{"arxiv_id":"2606.01161","last_updated":"2026-05-31T11:08:51Z","snapshot_observed_at":"2026-08-03T09:12:52.643122Z","submitted_at":"2026-05-31T11:08:51Z","title":"AcOrch: Accelerating Sampling-based GNN Training under CPU-NPU Heterogeneous Environments","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-06-28T16:37:20.774251Z"},"links":{"cited_paper":"/paper/2412.18106","citing_paper":"/paper/2606.01161"},"observation_digest":"sha256:1b804578da425c7e53eba263e3f449876b98ff646f8228cde4bfd2ab25230b05","observation_id":"948b4573-1367-45f1-adf1-ed2a3aaa1f87","resolution":{"observed_at":"2026-07-01T21:36:15.533193Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2412.18106/citation-record","integrity":"/paper/2412.18106/integrity","json":"/paper/2412.18106/citation-record.json","paper":"/paper/2412.18106"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:23.327427Z","title":"https://pytorch.org/blog/acceleratin g-llama3/?hss_channel=lcp-78618366/","venue":null,"work_id":"50ab4aa9-b98f-4a93-b49d-cc8c602a1c96","year":null},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.392066Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:7660a2ab94f9934a8f3af03db7c07636d7d120354304722bdd4b03ee6c6988dc","observation_id":"47cd4147-f3cc-47dd-ac60-4559b03ac1b3","resolution":{"observed_at":"2026-08-11T05:05:23.331066Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:23.315137Z","title":"https://flashinfer.ai/ 2024/02/02/introduce-flashinfer.html","venue":null,"work_id":"0851ae96-d046-417f-9e37-cb67b7bef513","year":2024},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.396611Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:c683584b89e864c1fc81375ee7b3387d90a4010aa8b9744ea738663ac9d19dc3","observation_id":"83a42cb8-8a3e-474b-a87f-8c1bf358989e","resolution":{"observed_at":"2026-08-11T05:05:23.319515Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:23.300551Z","title":"https://pytorch.org/blog/cutlass-p ing-pong-gemm-kernel/","venue":null,"work_id":"1623fbaa-be91-417c-a0e0-b1b20f1b879d","year":null},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.401104Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:3a837ac27298fecbfbce4b30381ac20313b3a5e104c311294fea94fd9d45cb03","observation_id":"22947872-6e5a-40b3-a0f1-7ad9f52a1758","resolution":{"observed_at":"2026-08-11T05:05:23.305150Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:23.287310Z","title":"https://huggingface.co/blog/layerskip","venue":null,"work_id":"1e70d6d1-0387-4fc7-b59d-f8a594b8814d","year":null},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.405940Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:4f14f1005ccc44bbd192de0161a5280d9b842746a3145ba8a21a2ce1b4a5c83b","observation_id":"a18affcc-3f87-4cba-b693-1485f0841d30","resolution":{"observed_at":"2026-08-11T05:05:23.291470Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:23.271778Z","title":"https: //pytorch.org/blog/flash-decoding/","venue":null,"work_id":"e16c97b7-8e22-4359-aeb2-8f7f9b11c969","year":null},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.410870Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:da856d9d670469e463d860483fe77b230fc5f66c8da39ae0e7106d1787f6c6b2","observation_id":"452c3868-d977-4ffc-a866-f90d86387830","resolution":{"observed_at":"2026-08-11T05:05:23.277652Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:23.259411Z","title":"Mirror of https://gitee.com/ascend/pytorch","venue":null,"work_id":"12dca552-2a97-41c3-a1e7-231fae66c915","year":null},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.415024Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:0f9dde9d5d6d97945b89a0f59d600d5814f700821a6e5e33daa952e037f15bee","observation_id":"64bc80c1-1cfc-46b9-aa7a-daf962464996","resolution":{"observed_at":"2026-08-11T05:05:23.263812Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:23.241589Z","title":"https://github.com/p ybind/pybind11","venue":null,"work_id":"240b45d7-f033-4d2d-8913-2b40c3ddf21b","year":null},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.419398Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:38efd7d92855f5c0216530aeede1ce741f2851e709d9718447ae500142687ac0","observation_id":"dd28661e-e65c-4227-9cc1-8c37acf99bae","resolution":{"observed_at":"2026-08-11T05:05:23.250528Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:23.227903Z","title":"https://docs.vllm.ai/e n/latest/automatic_prefix_caching/apc.html","venue":null,"work_id":"2feef6f1-e5d3-403e-b229-8ea42686c9b5","year":null},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.423136Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:b9f5643891b38445e04de36794ef531c67e43ba23a06f92dd772553defce1d7b","observation_id":"5edfda8b-cdd9-4bac-a27d-9644b9eb8474","resolution":{"observed_at":"2026-08-11T05:05:23.232533Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:23.214587Z","title":"https://www.hiascend.com/docum ent/detail/en/canncommercial/700/modeldevpt /ptmigr/ptaoplist_000006.html","venue":null,"work_id":"45af461a-fca0-42da-9b78-150a849a8d4a","year":null},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.426276Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:4e9e8a46c9908b07f5e1fb8a3cb14df6f7a548bac211324293d54b5382bc40a8","observation_id":"83cefb15-b204-4231-81d8-6abdf8fcb6ae","resolution":{"observed_at":"2026-08-11T05:05:23.219157Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:23.202568Z","title":"https://www.hiascend.com/doc_center/source /zh/Pytorch/60RC2/apiref/apilist/ptaoplist _000787.html","venue":null,"work_id":"5d76c5fa-aee2-4a83-aec9-f1b2de11d687","year":null},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.429593Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:816e2013d45317bb910cb6b5baebb0f51afe0b25e512bdc0b44be23a261cb759","observation_id":"111339f5-494f-42d3-99f4-e97c7f379713","resolution":{"observed_at":"2026-08-11T05:05:23.206340Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:23.191159Z","title":"https://www","venue":null,"work_id":"bfdc85d5-fa66-4ff1-bc1e-819efaba6a0f","year":null},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.434016Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:60ba539e412ecf2f9997808d7c33988add23160f53ebadf5334ea7ffb454d351","observation_id":"9e869776-9d22-46e9-b9b5-733fc3a691ac","resolution":{"observed_at":"2026-08-11T05:05:23.194461Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:23.178762Z","title":"https://www.hiascend.com/doc_center/sour ce/zh/CANNCommunityEdition/80RC1alpha001/ap iref/fmkadptapi/ptaoplist_000142.html","venue":null,"work_id":"7ecf1fd1-8160-44d6-bfe5-765e9629a384","year":null},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.437889Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:b45d26931f8e6eba627a86a71f344bafd57368280fb63c4db4be4be86e586080","observation_id":"f37ccd31-1e6d-4ad5-a445-3b98d387190f","resolution":{"observed_at":"2026-08-11T05:05:23.183067Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:23.165605Z","title":"https://github.com/v llm-project/vllm/tree/main/.buildkite/nig htly-benchmarks","venue":null,"work_id":"6ddaa0ad-86cf-4e56-89df-459b99a92cdc","year":null},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.441379Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:9e3da7df1271950b75913f4427476dffe3f1481c675893b17e3501b09a7fe03d","observation_id":"0bbb9e4b-c871-4b17-9262-7d6c6e8aa49e","resolution":{"observed_at":"2026-08-11T05:05:23.169997Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:23.149537Z","title":"https://github.com /vllm-project/vllm/pull/8054","venue":null,"work_id":"072ab6de-7c8c-46bc-be5c-79936e0ba927","year":null},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.445080Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:c142273787c3e147d6b6d58aa0f8174ec338e9d23cb03ab564810035d0129350","observation_id":"f099ea11-6cec-4a5b-a15d-cbcf46a99a58","resolution":{"observed_at":"2026-08-11T05:05:23.155090Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:23.134090Z","title":"https://developer.nvidia.com/blog/optimi zing-compute-shaders-for-l2-locality-using -thread-group-id-swizzling/ , July 2020","venue":null,"work_id":"c7a1c479-ece7-4019-97b1-d9fe98955cb3","year":2020},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.448986Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:5ddd46ec31769671bbe4ec48c920eb84f290521a609ea1b27305832be79657cd","observation_id":"6e76a983-1087-48a1-bdbd-d0c58b75f3c2","resolution":{"observed_at":"2026-08-11T05:05:23.139298Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:22.453803Z","title":"Mnemosyne: Parallelization strategies for efficiently serving multi-million context length llm in- ference requests without approximations","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.453803Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:01c3caa1931c71196a81ce3400bb34910a7337417b3d1422a9ad870964c08237","observation_id":"6b08a2df-99af-4b74-a82c-04bb52b56086","resolution":{"observed_at":"2026-08-11T05:05:22.453803Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:23.120476Z","title":"Taming throughput- latency tradeoff in llm inference with sarathi-serve","venue":null,"work_id":"ef9f5eaa-12ae-43bf-afa3-d770800dc889","year":2024},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.458077Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:e381c68c25b1619c29731777247cff9b168255e8e9d4eea1450f1db1440e0497","observation_id":"2d69371e-fbdb-400e-bda8-e7a15a12b8f2","resolution":{"observed_at":"2026-08-11T05:05:23.125471Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-06T15:43:00.292272Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-11T05:05:22.462957Z","title":"Sarathi: Efficient llm inference by piggy- backing decodes with chunked prefills","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.462957Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:acc840190ebe11e262d34d19b34137fba75b68f9f9058b1193a40ac2ea918898","observation_id":"36bab499-f5d5-491e-8201-f8db57f5b7b7","resolution":{"observed_at":"2026-08-11T05:05:22.462957Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:23.108468Z","title":null,"venue":null,"work_id":"7ee20b50-a6eb-46a7-841e-6e2255a8cda1","year":2020},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.468538Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:1160810aacffa4b22a291261e343cd276e2127f7165e05ab5bf9d5046666b71f","observation_id":"0d67471d-a83d-4795-9981-14a0fe5af7a9","resolution":{"observed_at":"2026-08-11T05:05:23.112597Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.10774","last_updated":"2024-06-14T23:32:32Z","snapshot_observed_at":"2026-07-06T17:17:56.276857Z","submitted_at":"2024-01-19T15:48:40Z","title":"Medusa: Simple LLM Inference Acceleration Framework with Multiple Decoding Heads","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.10774","snapshot_observed_at":"2026-08-11T05:05:22.473831Z","title":"Medusa: Simple llm inference acceleration frame- work with multiple decoding heads, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.473831Z"},"links":{"cited_paper":"/paper/2401.10774","citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:7bc3ee2216900aa1a0d8c8fbd366addca946cd1053647057e3427426c7cf3549","observation_id":"75bfec98-92e3-4eb8-8d29-2d43e65f6b38","resolution":{"observed_at":"2026-08-11T05:05:22.473831Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:22.478313Z","title":"End-to-end object detection with transform- ers","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.478313Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:9984ec651d3da585ca587827d3c904213448c229a0801d17b65654c38a6f43c0","observation_id":"5e8a7e0a-e400-4e19-943b-a7ad5eea69ee","resolution":{"observed_at":"2026-08-11T05:05:22.478313Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.01318","last_updated":"2023-02-02T18:44:11Z","snapshot_observed_at":"2026-08-10T21:52:49.568983Z","submitted_at":"2023-02-02T18:44:11Z","title":"Accelerating Large Language Model Decoding with Speculative Sampling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.01318","snapshot_observed_at":"2026-08-11T05:05:22.482679Z","title":"Accelerating large language model decoding with specu- lative sampling","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.482679Z"},"links":{"cited_paper":"/paper/2302.01318","citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:bca01f100173cab092b96355ad815aecb18004e9fd79a58a95ae976788903bd9","observation_id":"733cd9e4-6420-4366-b0d2-338805b786a5","resolution":{"observed_at":"2026-08-11T05:05:22.482679Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.08691","last_updated":"2023-07-17T17:50:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-17T17:50:36Z","title":"FlashAttention-2: Faster Attention with Better Parallelism and Work Partitioning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.08691","snapshot_observed_at":"2026-08-11T05:05:22.486482Z","title":"Flashattention-2: Faster attention with better parallelism and work partitioning (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.486482Z"},"links":{"cited_paper":"/paper/2307.08691","citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:cb78e2a94645996b1336a2cd0e9cb103224ab47de951d649debd0a74cc31e208","observation_id":"d0b6a347-0352-4d1d-8e03-58225f182926","resolution":{"observed_at":"2026-08-11T05:05:22.486482Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:23.087379Z","title":"Flashattention: Fast and memory- efficient exact attention with io-awareness","venue":null,"work_id":"a3fb93d8-3bc7-4585-aedd-400b3552b931","year":2022},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.490605Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:75828f19aeea4e15c45d8d109106b66f6292bfb87aab8cbfd81253cdd3de1abe","observation_id":"be8e9371-9f9c-4480-b7d6-75c73f083ede","resolution":{"observed_at":"2026-08-11T05:05:23.091824Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:23.073913Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale","venue":null,"work_id":"851ef19a-bbe1-409b-bfde-8629def2e561","year":2021},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.494447Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:aded18f96a80e02faff2c12d9514d0340f4ff2061c412e9b704f8a07c6a62490","observation_id":"a0bb6f57-3fb0-440d-a1ef-f3fa95263ba5","resolution":{"observed_at":"2026-08-11T05:05:23.078223Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.08671","last_updated":"2024-01-09T06:49:40Z","snapshot_observed_at":"2026-07-06T17:16:25.682072Z","submitted_at":"2024-01-09T06:49:40Z","title":"DeepSpeed-FastGen: High-throughput Text Generation for LLMs via MII and DeepSpeed-Inference","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.08671","snapshot_observed_at":"2026-08-11T05:05:22.498348Z","title":"Deepspeed-fastgen: High-throughput text generation for llms via mii and deepspeed-inference","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.498348Z"},"links":{"cited_paper":"/paper/2401.08671","citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:596f3072775d1d2073af229d443dda26e87f54db7b62cd37df9e2d7ac2bfc4fa","observation_id":"5c526a61-35fc-4073-9207-c18d2d746ee0","resolution":{"observed_at":"2026-08-11T05:05:22.498348Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.11181","last_updated":"2024-01-20T09:43:36Z","snapshot_observed_at":"2026-08-05T23:35:16.350557Z","submitted_at":"2024-01-20T09:43:36Z","title":"Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.11181","snapshot_observed_at":"2026-08-11T05:05:22.503397Z","title":"Inference without interfer- ence: Disaggregate llm inference for mixed downstream workloads","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.503397Z"},"links":{"cited_paper":"/paper/2401.11181","citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:eddf61f6ca81387341ed4ea47d65decebbbe714d0377a348f224b79c92fd3b17","observation_id":"ac97fea9-8f50-4cd6-a9b7-c55ae0c2bb51","resolution":{"observed_at":"2026-08-11T05:05:22.503397Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.08147","last_updated":"2024-08-15T13:32:25Z","snapshot_observed_at":"2026-07-06T19:01:06.124684Z","submitted_at":"2024-08-15T13:32:25Z","title":"P/D-Serve: Serving Disaggregated Large Language Model at Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.08147","snapshot_observed_at":"2026-08-11T05:05:22.508469Z","title":"P/d-serve: Serving disag- gregated large language model at scale","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.508469Z"},"links":{"cited_paper":"/paper/2408.08147","citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:9968a719accada9bc0d5491e4e2601c472b59c395fafcf3c0e709bf8f30c988a","observation_id":"9857202e-31fb-4f66-baee-4a8f4072702d","resolution":{"observed_at":"2026-08-11T05:05:22.508469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.18038","last_updated":"2025-02-16T18:09:25Z","snapshot_observed_at":"2026-07-06T19:38:35.617923Z","submitted_at":"2024-10-23T17:06:56Z","title":"POD-Attention: Unlocking Full Prefill-Decode Overlap for Faster LLM Inference","version":2},"cited_work":{"arxiv_id":"2410.18038","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.18038","snapshot_observed_at":"2026-08-11T05:05:22.769398Z","title":"POD-Attention: Unlocking Full Prefill-Decode Overlap for Faster LLM Inference","venue":"cs.LG","work_id":"1bf112b9-edc8-46f9-b53b-973653982977","year":2024},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.514760Z"},"links":{"cited_paper":"/paper/2410.18038","citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:3b2c3dbc30d03f33dbd63e6ab39dcc021741a6a77d29c390188814290b37ccd9","observation_id":"b7800d6c-7386-4453-a1c9-cd79dd7fe34b","resolution":{"observed_at":"2026-08-11T05:05:22.775442Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2001.08361","last_updated":"2020-01-23T03:59:20Z","snapshot_observed_at":"2026-07-06T08:52:12.656082Z","submitted_at":"2020-01-23T03:59:20Z","title":"Scaling Laws for Neural Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2001.08361","snapshot_observed_at":"2026-08-11T05:05:22.519070Z","title":"Scal- ing laws for neural language models","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.519070Z"},"links":{"cited_paper":"/paper/2001.08361","citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:8e602e07b875246ecbdb08413f7357d652c79d7fb62b8ade31299bd9f7a50a98","observation_id":"27cd03d3-196d-403b-b467-27393fa9a00e","resolution":{"observed_at":"2026-08-11T05:05:22.519070Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:22.523327Z","title":"Efficient memory man- agement for large language model serving with page- dattention","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.523327Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:36685ce7cdb9b16cb5da54dbf0532775fb12c898f438b7b43fd078bccef30654","observation_id":"bb016a02-9b7e-42f5-84f8-f03d8a495ba2","resolution":{"observed_at":"2026-08-11T05:05:22.523327Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:22.527297Z","title":"Fast inference from transformers via speculative decoding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.527297Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:767d70e9f007e24901929df75e58173052b12db51d4e5e8201f4d2918c075a0f","observation_id":"81fc6725-0bf6-436b-b36f-2d510aa3f4b4","resolution":{"observed_at":"2026-08-11T05:05:22.527297Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.15077","last_updated":"2025-03-04T13:58:39Z","snapshot_observed_at":"2026-08-03T09:40:31.365295Z","submitted_at":"2024-01-26T18:59:01Z","title":"EAGLE: Speculative Sampling Requires Rethinking Feature Uncertainty","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.15077","snapshot_observed_at":"2026-08-11T05:05:22.531342Z","title":"Eagle: Speculative sampling requires rethinking feature uncertainty","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.531342Z"},"links":{"cited_paper":"/paper/2401.15077","citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:129b29a181cca5ae6ab7910622c02d048173cccd0d875b7934e2496f07f200c6","observation_id":"b1aec2a8-ca3b-4f6d-9fb9-70510bc93d1e","resolution":{"observed_at":"2026-08-11T05:05:22.531342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:23.042756Z","title":"Ascend: a scalable and unified architecture for ubiquitous deep neural network computing: Industry track paper","venue":null,"work_id":"f4c8c690-8896-4386-ac38-f827a7493ea8","year":2021},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.535557Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:a39e0c96571ce1442bda158417bfd997ebe68245bed8ab7c8c99d361d051dd5d","observation_id":"106ad724-c428-45ed-a10e-76a854dcaa36","resolution":{"observed_at":"2026-08-11T05:05:23.047071Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:23.028376Z","title":"Davinci: A scalable architecture for neural network com- puting","venue":null,"work_id":"8420e1c0-de1f-46cb-baec-a3c87167f8bf","year":2019},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.539706Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:205a75789b2cbca49b5c5c12406abbfa6cb41406dad1a2ed474da9e2a0b4842f","observation_id":"07c0b696-aace-4bae-87a8-e6a60384a8ef","resolution":{"observed_at":"2026-08-11T05:05:23.033583Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.16663","last_updated":"2024-10-22T03:29:33Z","snapshot_observed_at":"2026-08-07T00:29:14.945462Z","submitted_at":"2024-10-22T03:29:33Z","title":"FastAttention: Extend FlashAttention2 to NPUs and Low-resource GPUs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.16663","snapshot_observed_at":"2026-08-11T05:05:22.543764Z","title":"Fastattention: Extend flashat- tention2 to npus and low-resource gpus","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.543764Z"},"links":{"cited_paper":"/paper/2410.16663","citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:be4e4ad266e22db9612a2a91764b64feb1d435c3f0e0016cd1dea76fff7e0ade","observation_id":"bfec7400-f702-40ed-9ca1-0d7ab176d340","resolution":{"observed_at":"2026-08-11T05:05:22.543764Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14066","last_updated":"2025-07-27T03:20:41Z","snapshot_observed_at":"2026-07-06T18:34:03.037351Z","submitted_at":"2024-06-20T07:43:33Z","title":"TurboSpec: Closed-loop Speculation Control System for Optimizing LLM Serving Goodput","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.14066","snapshot_observed_at":"2026-08-11T05:05:22.548112Z","title":"Optimizing specula- tive decoding for serving large language models using goodput","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.548112Z"},"links":{"cited_paper":"/paper/2406.14066","citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:8f3c70c08903dfd2b1c1843bf9e032f8446bca114a0ed69f9ea885621c84dd6a","observation_id":"bd49f4a0-a2f4-4f8f-9570-d8fc9bfaabf2","resolution":{"observed_at":"2026-08-11T05:05:22.548112Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:23.012649Z","title":"Specinfer: Accelerating large language model serving with tree-based speculative inference and verification","venue":null,"work_id":"86402cce-8edf-41b6-ac4d-83f4409886aa","year":2024},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.553178Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:ead3dfa7bd81c0e1ac363c2c6bb4485c9ad371d90c0c4f349b5a8121add2011c","observation_id":"e8651dbe-77c5-4aa3-9ffd-a011c769b9e4","resolution":{"observed_at":"2026-08-11T05:05:23.018073Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-08-07T12:08:57.217593Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-11T05:05:22.557261Z","title":"Moon- cake: A kvcache-centric disaggregated architecture for llm serving","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.557261Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:329605c625c78b886f40064e4977d5ce678151743ebd7ed8c161ef9d48367d2b","observation_id":"3fb187e3-ee08-4c7d-8e13-2b351b54ce6c","resolution":{"observed_at":"2026-08-11T05:05:22.557261Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.08608","last_updated":"2024-07-12T22:15:02Z","snapshot_observed_at":"2026-07-06T18:44:53.587276Z","submitted_at":"2024-07-11T15:44:48Z","title":"FlashAttention-3: Fast and Accurate Attention with Asynchrony and Low-precision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.08608","snapshot_observed_at":"2026-08-11T05:05:22.561973Z","title":"Flashattention- 3: Fast and accurate attention with asynchrony and low- precision","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.561973Z"},"links":{"cited_paper":"/paper/2407.08608","citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:51d1e1f673d1cede845d3f749c5492e686b6021cde6d0ff04b256e5a3f4058e2","observation_id":"b8c36028-98ec-4624-a5cc-0d7b426b321d","resolution":{"observed_at":"2026-08-11T05:05:22.561973Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-11T05:05:22.567647Z","title":"Llama 2: Open foundation and fine-tuned chat models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.567647Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:11f300ae525e5bafc9459e9e29b92377b3253d7745f377a8bffe9737b4fd490b","observation_id":"1f0dc711-fbd7-432d-be49-72d79524c17a","resolution":{"observed_at":"2026-08-11T05:05:22.567647Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10671","last_updated":"2024-09-10T13:25:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-15T12:35:42Z","title":"Qwen2 Technical Report","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10671","snapshot_observed_at":"2026-08-11T05:05:22.572111Z","title":"Qwen2 technical report","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.572111Z"},"links":{"cited_paper":"/paper/2407.10671","citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:b19470b6874303d25e9c9fe61a31722cb6b100dd5323857a6037f69ac44aec49","observation_id":"5b8170ae-176f-48f0-9769-8c7a0df895a6","resolution":{"observed_at":"2026-08-11T05:05:22.572111Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:22.997085Z","title":"ChunkAt- tention: Efficient self-attention with prefix-aware KV cache and two-phase partition","venue":null,"work_id":"4b9c2f77-e4cb-401c-b788-cdbb98e0cbaf","year":2024},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.576332Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:fa37bf17be4e94db8bcfbde09e68b4f0b1b2e65271de67c36623860d0ee74de5","observation_id":"76612cb5-28c6-48c6-951e-65d09d21bbce","resolution":{"observed_at":"2026-08-11T05:05:23.002281Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:22.983119Z","title":"Orca: A distributed serving system for transformer-based generative mod- els","venue":null,"work_id":"c21a90ca-6b4b-46ed-9e38-2a430817eae2","year":2022},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.580400Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:281502a02900fab22371f708d544fbf20ea3b25764e8201ec9901f461235bf85","observation_id":"d24a8bee-8606-495d-9c90-1a43c4577a8b","resolution":{"observed_at":"2026-08-11T05:05:22.987665Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.08168","last_updated":"2024-05-20T02:37:20Z","snapshot_observed_at":"2026-07-06T16:18:49.258642Z","submitted_at":"2023-09-15T05:34:32Z","title":"Draft & Verify: Lossless Large Language Model Acceleration via Self-Speculative Decoding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.08168","snapshot_observed_at":"2026-08-11T05:05:22.584734Z","title":"Draft & verify: Lossless large language model acceleration via self- speculative decoding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.584734Z"},"links":{"cited_paper":"/paper/2309.08168","citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:78681911d0cd16cc75e978e52ca0a738244e4076ffc0f345df83dc535560df31","observation_id":"8058edb3-d53b-41ee-a168-34bf101eebb0","resolution":{"observed_at":"2026-08-11T05:05:22.584734Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:22.969225Z","title":"Lookahead: An inference acceleration frame- work for large language model with lossless generation accuracy","venue":null,"work_id":"6b8952d6-125e-4147-97e4-d1d442aaa8bb","year":2024},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.589029Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:60dc3c07da7b11916e49829cadaf85e592c242a3ef783669149b9008a9b06714","observation_id":"7878538c-d35e-4ba6-93b8-b14f9a978001","resolution":{"observed_at":"2026-08-11T05:05:22.973937Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.07104","last_updated":"2024-06-06T00:10:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-12T09:34:27Z","title":"SGLang: Efficient Execution of Structured Language Model Programs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.07104","snapshot_observed_at":"2026-08-11T05:05:22.592491Z","title":"Sglang: Efficient execution of structured language model pro- grams, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.592491Z"},"links":{"cited_paper":"/paper/2312.07104","citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:fdfb63acf2882c7f42749fe298f45cd0dfe496eb196740c295c1dd161d80921f","observation_id":"794db7ca-1f52-48d4-af1f-04c218142f6b","resolution":{"observed_at":"2026-08-11T05:05:22.592491Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T05:05:22.955170Z","title":"Dist- serve: Disaggregating prefill and decoding for goodput- optimized large language model serving","venue":null,"work_id":"a032bda2-fda2-48de-8e3b-c45fb570a40f","year":2024},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.596292Z"},"links":{"citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:3e8f14b7cfca5d20a05bde52af7076e05eea8c641c87b4acb26c607c68b15836","observation_id":"c41de5c9-3c40-480d-8740-319d8f0be7ee","resolution":{"observed_at":"2026-08-11T05:05:22.960244Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-11T04:59:40.897729Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels"},"reference_resolution":{"displayed":48,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":22,"verified_exact":1,"verified_fuzzy":25},"total_outbound_references":48},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 48 of 48 outbound references and 1 inbound Pith citation observation for arXiv:2412.18106."}