{"as_of":"2026-08-19T22:48:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:be85f112f6d43c8e3a3d4aa208ff37c67deb45a822e94d40afc7d400634762e5","coverage":[{"denominator":54,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":54,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T11:57:01.881688Z","state":"measured"},{"denominator":60,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":60,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":6,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":6,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-30T21:47:00.295144Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-06-30T22:05:06.251185Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"cited_work":{"arxiv_id":"2505.03756","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.03756","snapshot_observed_at":"2026-06-30T22:05:06.251185Z","title":"Improving the serving performance of multi- LoRA large language models via efficient LoRA and KV cache management","venue":null,"work_id":"b1c11388-208f-43ae-9890-98b113e0a3f9","year":2025},"citing_paper":{"arxiv_id":"2604.06370","last_updated":"2026-04-07T18:52:25Z","snapshot_observed_at":"2026-08-15T13:00:48.586034Z","submitted_at":"2026-04-07T18:52:25Z","title":"ForkKV: Scaling Multi-LoRA Agent Serving via Copy-on-Write Disaggregated KV Cache","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-05-10T18:16:49.292491Z"},"links":{"cited_paper":"/paper/2505.03756","citing_paper":"/paper/2604.06370"},"observation_digest":"sha256:d9638b8a88445b1682eef5333021d6d4fc70d44b0247e7bd6467a32b4fc4d6e9","observation_id":"f7de1d06-576c-40c4-9f24-4a0533830e9c","resolution":{"observed_at":"2026-05-11T05:10:54.967305Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"cited_work":{"arxiv_id":"2505.03756","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.03756","snapshot_observed_at":"2026-06-30T22:05:06.251185Z","title":"Improving the serving performance of multi- LoRA large language models via efficient LoRA and KV cache management","venue":null,"work_id":"b1c11388-208f-43ae-9890-98b113e0a3f9","year":2025},"citing_paper":{"arxiv_id":"2604.07173","last_updated":"2026-04-08T15:01:04Z","snapshot_observed_at":"2026-08-15T06:11:31.394169Z","submitted_at":"2026-04-08T15:01:04Z","title":"InfiniLoRA: Disaggregated Multi-LoRA Serving for Large Language Models","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-10T17:23:40.872418Z"},"links":{"cited_paper":"/paper/2505.03756","citing_paper":"/paper/2604.07173"},"observation_digest":"sha256:db2006b728f229c473c480962fd2ab97aea88641de3ec2699a50191328e7acaf","observation_id":"73fcc33d-1d8e-4967-b5f1-28bf0cada3ec","resolution":{"observed_at":"2026-05-11T06:55:59.287808Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"cited_work":{"arxiv_id":"2505.03756","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.03756","snapshot_observed_at":"2026-06-30T22:05:06.251185Z","title":"Improving the serving performance of multi- LoRA large language models via efficient LoRA and KV cache management","venue":null,"work_id":"b1c11388-208f-43ae-9890-98b113e0a3f9","year":2025},"citing_paper":{"arxiv_id":"2604.16583","last_updated":"2026-04-17T14:34:53Z","snapshot_observed_at":"2026-08-17T05:40:24.551862Z","submitted_at":"2026-04-17T14:34:53Z","title":"POLAR: Online Learning for LoRA Adapter Caching and Routing in Edge LLM Serving","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-10T08:49:43.418576Z"},"links":{"cited_paper":"/paper/2505.03756","citing_paper":"/paper/2604.16583"},"observation_digest":"sha256:30b0cf7ff81e65815b4a13def4aa9de8fc451327af59949da3b8296b4b038332","observation_id":"17a2c2dc-19e2-408e-b1c8-aedd740e52ba","resolution":{"observed_at":"2026-05-10T08:53:04.269274Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"cited_work":{"arxiv_id":"2505.03756","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.03756","snapshot_observed_at":"2026-06-30T22:05:06.251185Z","title":"Improving the serving performance of multi- LoRA large language models via efficient LoRA and KV cache management","venue":null,"work_id":"b1c11388-208f-43ae-9890-98b113e0a3f9","year":2025},"citing_paper":{"arxiv_id":"2605.13779","last_updated":"2026-05-26T16:10:31Z","snapshot_observed_at":"2026-08-19T01:48:18.988681Z","submitted_at":"2026-05-13T16:59:08Z","title":"MinT: Managed Infrastructure for Training and Serving Millions of LLMs","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-14T19:25:12.407148Z"},"links":{"cited_paper":"/paper/2505.03756","citing_paper":"/paper/2605.13779"},"observation_digest":"sha256:7fadfea8a0e492ab1feca7e6de8e87da3fc0f3ce7a4b81442a91437805b29b10","observation_id":"4c19f3af-b35e-44ce-9b08-b052bb18a265","resolution":{"observed_at":"2026-05-14T19:27:51.818338Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"cited_work":{"arxiv_id":"2505.03756","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.03756","snapshot_observed_at":"2026-06-30T22:05:06.251185Z","title":"Improving the serving performance of multi- LoRA large language models via efficient LoRA and KV cache management","venue":null,"work_id":"b1c11388-208f-43ae-9890-98b113e0a3f9","year":2025},"citing_paper":{"arxiv_id":"2605.13779","last_updated":"2026-05-26T16:10:31Z","snapshot_observed_at":"2026-08-19T01:48:18.988681Z","submitted_at":"2026-05-13T16:59:08Z","title":"MinT: Managed Infrastructure for Training and Serving Millions of LLMs","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-30T21:47:00.295144Z"},"links":{"cited_paper":"/paper/2505.03756","citing_paper":"/paper/2605.13779"},"observation_digest":"sha256:3c00703515acea604a8822ce2e93715754bbfa5832ed6d686a58bb94f3e8c6aa","observation_id":"9980b588-d301-4673-af44-6c390837a1b1","resolution":{"observed_at":"2026-06-30T22:05:06.252808Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"cited_work":{"arxiv_id":"2505.03756","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.03756","snapshot_observed_at":"2026-06-30T22:05:06.251185Z","title":"Improving the serving performance of multi- LoRA large language models via efficient LoRA and KV cache management","venue":null,"work_id":"b1c11388-208f-43ae-9890-98b113e0a3f9","year":2025},"citing_paper":{"arxiv_id":"2605.14217","last_updated":"2026-05-14T00:19:41Z","snapshot_observed_at":"2026-08-14T02:13:33.557619Z","submitted_at":"2026-05-14T00:19:41Z","title":"PreFT: Prefill-only finetuning for efficient inference","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-05-15T02:10:14.721584Z"},"links":{"cited_paper":"/paper/2505.03756","citing_paper":"/paper/2605.14217"},"observation_digest":"sha256:3d17e5b0d8f10b5ff30bd367f253f97971ba8324e400b4af2237966ce4f8475a","observation_id":"d4ac3560-e4c6-4f14-a0bf-fe5cb1507393","resolution":{"observed_at":"2026-05-15T02:13:30.657350Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2505.03756/citation-record","integrity":"/paper/2505.03756/integrity","json":"/paper/2505.03756/citation-record.json","paper":"/paper/2505.03756"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-15T04:42:28.752204Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-16T11:57:01.692789Z","title":"Sarathi: Efficient llm inference by piggy- backing decodes with chunked prefills","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.692789Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:8b7db709c28f40cfd098ccec8bd49042047ab4ddd84bfa100d758b6971f715b8","observation_id":"d77a65af-c5f5-4158-8fbc-7d35818c28b1","resolution":{"observed_at":"2026-08-16T11:57:01.692789Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11514","last_updated":"2024-07-30T23:37:20Z","snapshot_observed_at":"2026-08-19T12:40:37.732356Z","submitted_at":"2023-12-12T18:57:08Z","title":"LLM in a flash: Efficient Large Language Model Inference with Limited Memory","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11514","snapshot_observed_at":"2026-08-16T11:57:01.697534Z","title":"Llm in a flash: Efficient large language model inference with lim- ited memory","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.697534Z"},"links":{"cited_paper":"/paper/2312.11514","citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:6c2b38e0582092f50cb92b2ce9004278d1ab18c674e28f66fbac164cdc864729","observation_id":"f34dadd4-cb6a-456b-a374-2f65acd35538","resolution":{"observed_at":"2026-08-16T11:57:01.697534Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:57:02.535412Z","title":"Introducing apple’s on-device and server foun- dation models, 2025","venue":null,"work_id":"9a1319f3-fbdb-4ae7-8449-ec29d3431e17","year":2025},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.701797Z"},"links":{"citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:1a3be30690eaa44faf54416a4a500f22bd53d2dcad80ebe532fae9fe26c96e09","observation_id":"70718582-7ee3-4eda-a965-d2cc5ed84ee7","resolution":{"observed_at":"2026-08-16T11:57:02.538980Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.08061","last_updated":"2023-08-15T22:26:58Z","snapshot_observed_at":"2026-08-17T16:25:59.634094Z","submitted_at":"2023-08-15T22:26:58Z","title":"The Costly Dilemma: Generalization, Evaluation and Cost-Optimal Deployment of Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.08061","snapshot_observed_at":"2026-08-16T11:57:01.705679Z","title":"The costly dilemma: generalization, evaluation and cost- optimal deployment of large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.705679Z"},"links":{"cited_paper":"/paper/2308.08061","citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:0a1a996c359677c50b607dc0595f72b1a1729ede1c05fdf9f9006aa66e3f9cc2","observation_id":"f4768bd2-f376-46a5-a399-07b5d37ae985","resolution":{"observed_at":"2026-08-16T11:57:01.705679Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:57:02.524657Z","title":"Taskmaster-1:toward a realistic and diverse dialog dataset","venue":null,"work_id":"b9c61201-da21-4c09-9ae4-f3a1bc453e22","year":2019},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.709512Z"},"links":{"citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:8442bf034346abdc34a213d9bb57709c0c64a3eafc25a5c230eda5d331b45deb","observation_id":"a72d59f2-eeb1-45fc-b5cb-bfa4892ebc69","resolution":{"observed_at":"2026-08-16T11:57:02.528286Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:57:02.513691Z","title":"Punica: Multi-tenant lora serving","venue":null,"work_id":"7be0d6ae-10fb-4432-9ca6-537d0431b93f","year":2024},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.713583Z"},"links":{"citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:b2f61a3c4293736063f1398a63b39516778bcd8fe3d364d8c41a3cd9bea36017","observation_id":"1410a725-0eb3-4e76-8d83-afebd3df86f7","resolution":{"observed_at":"2026-08-16T11:57:02.517409Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-16T11:57:01.717527Z","title":"Chatbot arena: An open platform for evaluating llms by human preference","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.717527Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:ba160db6f7d775b76e92e9f3174ca3cdfbdd91cdf2220de7b44c78517c28faf7","observation_id":"4c54425e-6112-41a9-9e66-d24d841ff385","resolution":{"observed_at":"2026-08-16T11:57:01.717527Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:57:01.721637Z","title":"Palm: Scaling language modeling with pathways","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.721637Z"},"links":{"citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:19153c53299898dc3eb626635eab2a72eab560a7f126bddf4087874fbf386c19","observation_id":"48f314f6-b18b-4715-9eae-df39be8b22a8","resolution":{"observed_at":"2026-08-16T11:57:01.721637Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:57:02.493820Z","title":"Introduction to tpus, 2023","venue":null,"work_id":"fe9362d1-0f71-48d2-9a5a-418560fe53e7","year":2023},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.724828Z"},"links":{"citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:c08b9f4d859fb26e4e408c7f17334598e9f3defbd5fed13bfe0aeff879b2a5d9","observation_id":"e5c3a67f-f606-4125-99ec-2ff937571ba5","resolution":{"observed_at":"2026-08-16T11:57:02.498115Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:57:02.482596Z","title":"sglang: A fast serving framework for large language models and vision language models., 2024","venue":null,"work_id":"08d85bf7-6206-468d-b771-e57dc573f69c","year":2024},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.728100Z"},"links":{"citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:f188620b4cf92d2d1e8af632433d67b99ade8b1b0c5ead2ab1314fdc920978ec","observation_id":"2b3a5b57-94c8-458f-8990-c692ddf520e8","resolution":{"observed_at":"2026-08-16T11:57:02.486403Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:57:02.471829Z","title":"High bandwidth memory, 2023","venue":null,"work_id":"fdf58ab8-737f-4cab-b12e-2aee0dd1d684","year":2023},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.731413Z"},"links":{"citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:c475157e09c19b18cfcb1b450ff550724b2257974464082d810bb470d70000ee","observation_id":"2fe70934-7afc-4584-a05b-49a4a7ce9db3","resolution":{"observed_at":"2026-08-16T11:57:02.475515Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:57:02.461089Z","title":"Trie, 2023","venue":null,"work_id":"524453d5-2214-46c3-bead-93d49e3846e1","year":2023},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.734605Z"},"links":{"citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:bb193961044acd3fa89baa9d68a568d44cd27d8bfa7123e406b5ed7be2416f7d","observation_id":"edc0a231-ccfd-4d35-84b1-a2034f778590","resolution":{"observed_at":"2026-08-16T11:57:02.464723Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:57:02.450517Z","title":"Nvidia a100 tensor core gpu, 2024","venue":null,"work_id":"f8690d27-34b0-4e5c-a657-61b0d414ae91","year":2024},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.738280Z"},"links":{"citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:18c3dd302d7415f3201bea4c434862cc414f6d373461b56500f849589dba1dcc","observation_id":"8e80c258-b45c-41fb-89d2-9f22bce18848","resolution":{"observed_at":"2026-08-16T11:57:02.454006Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:57:01.741620Z","title":"Qlora: Efficient finetuning of quan- tized llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.741620Z"},"links":{"citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:e6194abfe54c46ebb83ac3ca2ac03c7d2a7915531770383b6eaa7581d36bc116","observation_id":"4167d71a-fbd4-4219-9538-4efcdb01e611","resolution":{"observed_at":"2026-08-16T11:57:01.741620Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:57:01.744787Z","title":"Gpt-3: Its nature, scope, limits, and consequences","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.744787Z"},"links":{"citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:c4ee500045bbd92df10455700e0b260ba1e4cf63cf3cceb463442e1bb27a4a52","observation_id":"f39b5a8a-f1b8-4a92-a694-5d52b6a1857c","resolution":{"observed_at":"2026-08-16T11:57:01.744787Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.19708","last_updated":"2024-06-30T23:50:38Z","snapshot_observed_at":"2026-08-16T14:07:07.762606Z","submitted_at":"2024-03-23T10:42:49Z","title":"Cost-Efficient Large Language Model Serving for Multi-turn Conversations with CachedAttention","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.19708","snapshot_observed_at":"2026-08-16T11:57:01.748120Z","title":"Attentionstore: Cost-effective atten- tion reuse across multi-turn conversations in large lan- guage model serving","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.748120Z"},"links":{"cited_paper":"/paper/2403.19708","citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:cf9773a25a2756742cc3d1280739ec5ab7e94774676bcf17540ac49d2b2e54a0","observation_id":"5eb411b4-7279-4a1f-9c5d-93c1778190cd","resolution":{"observed_at":"2026-08-16T11:57:01.748120Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:57:02.425005Z","title":"Prompt cache: Modular attention reuse for low-latency inference","venue":null,"work_id":"23e83bd5-cbce-4beb-86a4-8e74d266c747","year":2024},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.751777Z"},"links":{"citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:6b8db42eb74530ddd3bfeb08227036fc0f4db80e0a7214d0eb76f28e241c59c9","observation_id":"b9486e9e-d77f-4486-85da-e34bb823c605","resolution":{"observed_at":"2026-08-16T11:57:02.428587Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.09685","last_updated":"2021-10-16T18:40:34Z","snapshot_observed_at":"2026-08-17T18:04:53.578114Z","submitted_at":"2021-06-17T17:37:18Z","title":"LoRA: Low-Rank Adaptation of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.09685","snapshot_observed_at":"2026-08-16T11:57:01.755128Z","title":"Lora: Low-rank adaptation of large language models","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.755128Z"},"links":{"cited_paper":"/paper/2106.09685","citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:82a6fd4ea293382d9d8376f670710ab953558eb191ba04527ba607fec02d7cbe","observation_id":"ae4ccf8e-dd1a-4b43-a95e-5831a7b1955a","resolution":{"observed_at":"2026-08-16T11:57:01.755128Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.13269","last_updated":"2024-08-19T03:31:19Z","snapshot_observed_at":"2026-08-16T15:13:34.132588Z","submitted_at":"2023-07-25T05:39:21Z","title":"LoraHub: Efficient Cross-Task Generalization via Dynamic LoRA Composition","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.13269","snapshot_observed_at":"2026-08-16T11:57:01.758544Z","title":"Lorahub: Efficient cross- task generalization via dynamic lora composition","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.758544Z"},"links":{"cited_paper":"/paper/2307.13269","citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:2853503d29c42e1883266c2a8de114b12dbbeee48ffb269203278530c93c1f6c","observation_id":"aa2c5c4d-d506-407e-adc0-556294f77415","resolution":{"observed_at":"2026-08-16T11:57:01.758544Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2411.17741","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:57:02.209501Z","title":"Chameleon: Adaptive caching and scheduling for many- adapter llm inference environments","venue":null,"work_id":"96933737-93a4-4374-86fa-6c296ac2f1c5","year":2024},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.762195Z"},"links":{"citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:937966d1ffa5874569796bb52315378b99db37c9787a6a431bcb01c17e1168df","observation_id":"e365f6ed-7408-49b3-a184-184c30990a42","resolution":{"observed_at":"2026-08-16T11:57:02.215502Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:57:01.765927Z","title":"Efficient memory man- agement for large language model serving with page- dattention","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.765927Z"},"links":{"citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:5a06f9f21e21704d6b71d8850c625897296c738e6d063282e79a2b96ed40c945","observation_id":"3c5753f6-b4ac-4728-bc25-464a14aa2bc9","resolution":{"observed_at":"2026-08-16T11:57:01.765927Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.08691","last_updated":"2021-09-02T17:34:41Z","snapshot_observed_at":"2026-08-16T20:01:36.160048Z","submitted_at":"2021-04-18T03:19:26Z","title":"The Power of Scale for Parameter-Efficient Prompt Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.08691","snapshot_observed_at":"2026-08-16T11:57:01.769214Z","title":"The power of scale for parameter-efficient prompt tuning","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.769214Z"},"links":{"cited_paper":"/paper/2104.08691","citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:6c9ab6baeab6cade7a631402f8c38264b3ae2076630344bc45c756f6f002c3c4","observation_id":"6cc91c4d-7915-4744-8885-301a44c8a8af","resolution":{"observed_at":"2026-08-16T11:57:01.769214Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.11240","last_updated":"2024-01-20T14:37:48Z","snapshot_observed_at":"2026-08-19T00:38:58.362291Z","submitted_at":"2024-01-20T14:37:48Z","title":"CaraServe: CPU-Assisted and Rank-Aware LoRA Serving for Generative LLM Inference","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.11240","snapshot_observed_at":"2026-08-16T11:57:01.772247Z","title":"Caraserve: Cpu-assisted and rank- aware lora serving for generative llm inference","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.772247Z"},"links":{"cited_paper":"/paper/2401.11240","citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:b3e58d8343260f70d4801345bebcec4a3d8a67e0484fbbe2288ae75c5b994af1","observation_id":"28475c05-f47c-42e4-969c-d3453d406675","resolution":{"observed_at":"2026-08-16T11:57:01.772247Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2101.00190","last_updated":"2021-01-01T08:00:36Z","snapshot_observed_at":"2026-08-12T12:37:28.207761Z","submitted_at":"2021-01-01T08:00:36Z","title":"Prefix-Tuning: Optimizing Continuous Prompts for Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2101.00190","snapshot_observed_at":"2026-08-16T11:57:01.775659Z","title":"Prefix-tuning: Optimiz- ing continuous prompts for generation","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.775659Z"},"links":{"cited_paper":"/paper/2101.00190","citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:986418a2d62fd665b9e82854d40a56081e28e33605b86577b86c5cec7d1b727f","observation_id":"7fff902a-b521-4faa-a448-88f410897a0b","resolution":{"observed_at":"2026-08-16T11:57:01.775659Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05459","last_updated":"2024-05-08T06:16:23Z","snapshot_observed_at":"2026-08-19T17:12:26.495549Z","submitted_at":"2024-01-10T09:25:45Z","title":"Personal LLM Agents: Insights and Survey about the Capability, Efficiency and Security","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05459","snapshot_observed_at":"2026-08-16T11:57:01.779138Z","title":"Personal llm agents: Insights and survey about the capability, efficiency and security","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.779138Z"},"links":{"cited_paper":"/paper/2401.05459","citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:b0c58f6c376f01d908cb2923035d41298bf146ed3fe675e333947349900879b1","observation_id":"13b5eff5-021e-4f5a-9f8e-6772002df314","resolution":{"observed_at":"2026-08-16T11:57:01.779138Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.09353","last_updated":"2024-07-09T05:59:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-14T17:59:34Z","title":"DoRA: Weight-Decomposed Low-Rank Adaptation","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.09353","snapshot_observed_at":"2026-08-16T11:57:01.782606Z","title":"Dora: Weight-decomposed low-rank adaptation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.782606Z"},"links":{"cited_paper":"/paper/2402.09353","citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:2d90765522f73bb99ed8edc64b11a9d53714ce3d7826c8c21e346dfadedd92d5","observation_id":"eb81c90b-306d-477c-9f2e-4d716beb127a","resolution":{"observed_at":"2026-08-16T11:57:01.782606Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:57:02.407491Z","title":"Instruct-tune llama on consumer hardware using alpaca-lora, 2023","venue":null,"work_id":"9b05867f-b3e3-4e89-9496-d170bb4c891f","year":2023},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.786129Z"},"links":{"citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:43027f5079be4ce06599d0e65915123ef9e92fc70377df075361d2a19867dc32","observation_id":"f0398a13-59c0-4c8b-be41-69a2b40332ce","resolution":{"observed_at":"2026-08-16T11:57:02.411273Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10627","last_updated":"2024-07-15T11:26:07Z","snapshot_observed_at":"2026-08-17T21:56:31.436746Z","submitted_at":"2024-07-15T11:26:07Z","title":"Arena Learning: Build Data Flywheel for LLMs Post-training via Simulated Chatbot Arena","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10627","snapshot_observed_at":"2026-08-16T11:57:01.789502Z","title":"Arena learning: Build data flywheel for llms post-training via simulated chatbot arena","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.789502Z"},"links":{"cited_paper":"/paper/2407.10627","citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:66ddd116b836c77c2f7ee90c4d77337e760864ce6ee8492cb6051e10c5849a02","observation_id":"eef52f8a-75f9-4197-9171-bbebf9c66ade","resolution":{"observed_at":"2026-08-16T11:57:01.789502Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2005.14165","last_updated":"2020-07-22T19:47:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-05-28T17:29:03Z","title":"Language Models are Few-Shot Learners","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.14165","snapshot_observed_at":"2026-08-16T11:57:01.793126Z","title":"Language models are few-shot learners","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.793126Z"},"links":{"cited_paper":"/paper/2005.14165","citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:118f1c8ee673609603b939aaa2b43f3a4cc991053752c6352faf729708d25328","observation_id":"26d9b191-8666-40f4-9cc5-16cb73f88a6b","resolution":{"observed_at":"2026-08-16T11:57:01.793126Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.01050","last_updated":"2024-08-02T06:56:59Z","snapshot_observed_at":"2026-08-16T13:29:02.359416Z","submitted_at":"2024-08-02T06:56:59Z","title":"The Impact of Hyperparameters on Large Language Model Inference Performance: An Evaluation of vLLM and HuggingFace Pipelines","version":1},"cited_work":{"arxiv_id":"2408.01050","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.01050","snapshot_observed_at":"2026-08-16T11:57:02.037419Z","title":"The Impact of Hyperparameters on Large Language Model Inference Performance: An Evaluation of vLLM and HuggingFace Pipelines","venue":"cs.SE","work_id":"47c8581f-4629-4a72-820a-381ea0ddbf49","year":2024},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.796567Z"},"links":{"cited_paper":"/paper/2408.01050","citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:7fe2a63726b03a5681d8d772600b59a78ff314e3cb248596533f4e0a64bc1120","observation_id":"3f6700a8-b82d-4592-a1b5-7e9fd585d7b7","resolution":{"observed_at":"2026-08-16T11:57:02.043357Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:57:02.396324Z","title":"Chatgpt, 2020","venue":null,"work_id":"c40603e5-1f82-48b3-a5eb-b5eb769777e5","year":2020},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.800176Z"},"links":{"citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:2dfc70b6b851400899abf88d6bc11a87b72662a7239bd4a3c88cb79944010f56","observation_id":"1a54de16-150e-42eb-a09b-f09f61482d29","resolution":{"observed_at":"2026-08-16T11:57:02.400603Z","resolver_source":"raw_fallback","status":"parse_uncertain"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:57:02.385151Z","title":"torch.stream — pytorch 2.0.1 documentation, 2023","venue":null,"work_id":"f3d157c1-a739-402f-b54e-dea28bb8bcb4","year":2023},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.803652Z"},"links":{"citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:1c712b71a8ac030fb6a5570dd74b3d847ccb096ca73e74fde6f7bb69ed062598","observation_id":"e426f054-1e11-4d0a-a0cb-c532dd4c1ebf","resolution":{"observed_at":"2026-08-16T11:57:02.389230Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:57:02.374172Z","title":"Moon- cake: Kimi’s kvcache-centric architecture for llm serv- ing","venue":null,"work_id":"dbd12624-257d-4f34-8b7f-5e27e3c527a2","year":2024},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.808007Z"},"links":{"citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:cf3ee41d6a1fd9c3c131d4117a7fc8847820360dee8a38d872c7cfb36b7b35af","observation_id":"bd27a102-1e9d-4a79-bff1-db2cc4a90ac6","resolution":{"observed_at":"2026-08-16T11:57:02.377850Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:57:02.362939Z","title":"Serverless in the wild: Characterizing and optimizing the serverless workload at a large cloud provider","venue":null,"work_id":"bc3a4060-b4e0-4c60-b035-c4b996d8c4f4","year":2020},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.811318Z"},"links":{"citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:bbe8f2415d578217d9e8599b2983d100cf3043c9928ed7d3f3a13c7ee0b232f0","observation_id":"e806169b-dbc6-4605-81f9-914cbc8931d9","resolution":{"observed_at":"2026-08-16T11:57:02.366647Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1911.02150","last_updated":"2019-11-06T00:19:05Z","snapshot_observed_at":"2026-07-06T08:35:01.386074Z","submitted_at":"2019-11-06T00:19:05Z","title":"Fast Transformer Decoding: One Write-Head is All You Need","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1911.02150","snapshot_observed_at":"2026-08-16T11:57:01.815067Z","title":"Fast transformer decoding: One write- head is all you need","venue":null,"work_id":null,"year":1911},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.815067Z"},"links":{"cited_paper":"/paper/1911.02150","citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:f35a79df2525c20d225da10c19d25d298c2cc4b0c1ab84a3aa1d78e096615290","observation_id":"53bc2ed7-fd1f-43e7-ac63-079b8163d743","resolution":{"observed_at":"2026-08-16T11:57:01.815067Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:57:02.352012Z","title":"Slora: Scalable serving of thousands of lora adapters","venue":null,"work_id":"0ad066e1-0f83-4cd1-b132-9815ae59f897","year":2024},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.818586Z"},"links":{"citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:3824ae7c851f7e82849282e71f830acd418325434d9ffbb83148a463df68a562","observation_id":"f2197089-74ee-4b9d-8be4-289cfa67e394","resolution":{"observed_at":"2026-08-16T11:57:02.355674Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.09586","last_updated":"2019-09-12T15:44:51Z","snapshot_observed_at":"2026-08-19T08:13:23.558641Z","submitted_at":"2019-09-12T15:44:51Z","title":"Understanding LSTM -- a tutorial into Long Short-Term Memory Recurrent Neural Networks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.09586","snapshot_observed_at":"2026-08-16T11:57:01.822147Z","title":"Understanding lstm–a tutorial into long short-term memory recurrent neural networks","venue":null,"work_id":null,"year":1909},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.822147Z"},"links":{"cited_paper":"/paper/1909.09586","citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:2c6f3b2ac678a9f7a65d00b51da6f78185bd871ac8b73fe5f71a879459c6d85b","observation_id":"29180fcf-06f0-49f7-b4fb-cbbf98c2855d","resolution":{"observed_at":"2026-08-16T11:57:01.822147Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-16T11:57:01.825709Z","title":"Llama 2: Open foundation and fine-tuned chat models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.825709Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:e817a346d220092b51d05fe9b94ecb261cdd6706176a13d70aaab5e69b5c86a0","observation_id":"6bfd6808-f4ef-406d-9234-86793841d94b","resolution":{"observed_at":"2026-08-16T11:57:01.825709Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:57:02.341191Z","title":"Discovering finance keywords via continuous-space lan- guage models","venue":null,"work_id":"10e7c63d-3721-4262-81b0-01e7c77f1139","year":2016},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.828982Z"},"links":{"citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:39243c2e8feb0f2c8f2e2609f295dd7299d8088de51bd881f6ea8f2530e4e3a6","observation_id":"de06853e-ebe8-446f-b414-d009d559bd45","resolution":{"observed_at":"2026-08-16T11:57:02.344777Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:57:02.329490Z","title":"vllm: A high-throughput and memory-efficient inference and serving engine for llms","venue":null,"work_id":"428b5b17-09b4-410e-b0c7-e1dba0e745da","year":null},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.832167Z"},"links":{"citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:2f31a5849c3535d2a5d9629d8ac124b4f444cee40764b8be151092b35e794be2","observation_id":"71c9a847-92c9-4836-9e38-d7dcd8350fd1","resolution":{"observed_at":"2026-08-16T11:57:02.333594Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.18242","last_updated":"2025-03-22T09:29:15Z","snapshot_observed_at":"2026-08-17T22:14:24.262905Z","submitted_at":"2024-07-25T17:57:12Z","title":"LoRA-Pro: Are Low-Rank Adapters Properly Optimized?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.18242","snapshot_observed_at":"2026-08-16T11:57:01.835511Z","title":"Lora-pro: Are low-rank adapters properly optimized? arXiv preprint arXiv:2407.18242, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.835511Z"},"links":{"cited_paper":"/paper/2407.18242","citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:61bd1b5a48dd1cd9c9aaea3d38e751265d39f692e1b21a5bd2b6de696e5962c0","observation_id":"b0b5f971-9565-49b9-a289-803c287750b0","resolution":{"observed_at":"2026-08-16T11:57:01.835511Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:57:02.318191Z","title":"{dLoRA}: Dynamically or- chestrating requests and adapters for{LoRA}{LLM} serving","venue":null,"work_id":"1c84b58f-0f01-4030-9ca5-2525cbdd8bc2","year":2024},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.839142Z"},"links":{"citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:8663faf9551398e545e76fa26907cb95cdb4580bdd7d7f153b74f79d8f8336a1","observation_id":"c4704b4f-dfea-4f00-bff9-862ee3e83cee","resolution":{"observed_at":"2026-08-16T11:57:02.321911Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.16444","last_updated":"2025-04-03T22:49:22Z","snapshot_observed_at":"2026-08-16T13:49:22.714428Z","submitted_at":"2024-05-26T06:00:17Z","title":"CacheBlend: Fast Large Language Model Serving for RAG with Cached Knowledge Fusion","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.16444","snapshot_observed_at":"2026-08-16T11:57:01.842609Z","title":"Cacheblend: Fast large language model serving with cached knowledge fusion","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.842609Z"},"links":{"cited_paper":"/paper/2405.16444","citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:76bca632ac7e0523b8c959d12808427a9932eee0edd2573815164c501ba44f68","observation_id":"bd84af8d-a272-4a4e-8be5-ebd876a11f20","resolution":{"observed_at":"2026-08-16T11:57:01.842609Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:57:01.846060Z","title":"Orca: A distributed serving system for transformer-based generative mod- els","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.846060Z"},"links":{"citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:b0e5a303be2c941cb6d62e51133ac019c3bb565e4ec549ade71cde5e48e83c19","observation_id":"5f863be8-a115-493b-86e9-1b1388c262ce","resolution":{"observed_at":"2026-08-16T11:57:01.846060Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.05516","last_updated":"2024-10-07T17:21:57Z","snapshot_observed_at":"2026-08-17T16:57:48.625848Z","submitted_at":"2023-12-09T09:55:07Z","title":"Stateful Large Language Model Serving with Pensieve","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.05516","snapshot_observed_at":"2026-08-16T11:57:01.849601Z","title":"Stateful large language model serving with pensieve","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.849601Z"},"links":{"cited_paper":"/paper/2312.05516","citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:1a8b1644784e331b3948d1f8c2465cacbde7606edf8e3afd81f089ae1d020867","observation_id":"4c247a40-3f34-474e-9abf-c3d79632cc3c","resolution":{"observed_at":"2026-08-16T11:57:01.849601Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2004.11867","last_updated":"2020-04-24T17:21:32Z","snapshot_observed_at":"2026-08-13T22:54:50.544752Z","submitted_at":"2020-04-24T17:21:32Z","title":"Improving Massively Multilingual Neural Machine Translation and Zero-Shot Translation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2004.11867","snapshot_observed_at":"2026-08-16T11:57:01.853284Z","title":"Improving massively multilingual neural machine translation and zero-shot translation","venue":null,"work_id":null,"year":2004},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.853284Z"},"links":{"cited_paper":"/paper/2004.11867","citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:452b0f18df9b1837d5e6d573ef368cff828cf3f29adcaf451bf6e57d613d3fa3","observation_id":"4a2844c1-3634-45de-b2a6-b6c669d62409","resolution":{"observed_at":"2026-08-16T11:57:01.853284Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1609.08144","last_updated":"2016-10-08T19:10:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2016-09-26T19:59:55Z","title":"Google's Neural Machine Translation System: Bridging the Gap between Human and Machine Translation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1609.08144","snapshot_observed_at":"2026-08-16T11:57:01.857107Z","title":"Google’s neural machine translation system: Bridging the gap between human and machine transla- tion","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.857107Z"},"links":{"cited_paper":"/paper/1609.08144","citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:5e5792247a306b45a9b19782946c15a5a67717ee7d6c1d1135b357414a373718","observation_id":"704ea79e-3b71-4257-89f0-78163f6d5e8b","resolution":{"observed_at":"2026-08-16T11:57:01.857107Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.10512","last_updated":"2023-12-20T20:56:14Z","snapshot_observed_at":"2026-08-12T23:46:47.414754Z","submitted_at":"2023-03-18T22:36:25Z","title":"AdaLoRA: Adaptive Budget Allocation for Parameter-Efficient Fine-Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.10512","snapshot_observed_at":"2026-08-16T11:57:01.860891Z","title":"Adalora: Adaptive bud- get allocation for parameter-efficient fine-tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.860891Z"},"links":{"cited_paper":"/paper/2303.10512","citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:f9f784a37868aa3461430a53ff1d75be6302596693a876bc2e68f83322543a26","observation_id":"111a0041-f61d-4710-8eb1-2b8f4258e3c5","resolution":{"observed_at":"2026-08-16T11:57:01.860891Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:57:01.864874Z","title":"Faster and cheaper serverless computing on harvested resources","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.864874Z"},"links":{"citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:7017359bb0ea09364fa06e843832db177092e3d946c1c996ec801192a36060b5","observation_id":"2ea5a426-71ac-40ec-8787-d347aea43be1","resolution":{"observed_at":"2026-08-16T11:57:01.864874Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00732","last_updated":"2024-04-29T04:01:45Z","snapshot_observed_at":"2026-08-16T13:57:06.445882Z","submitted_at":"2024-04-29T04:01:45Z","title":"LoRA Land: 310 Fine-tuned LLMs that Rival GPT-4, A Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00732","snapshot_observed_at":"2026-08-16T11:57:01.868044Z","title":"Lora land: 310 fine-tuned llms that rival gpt-4, a technical report","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.868044Z"},"links":{"cited_paper":"/paper/2405.00732","citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:6344063ec4e77ad3603a7eac5ff36abad2e0328467313fe4b8b0108831436eed","observation_id":"af570e82-87f7-43d2-a34f-3514cbb703b2","resolution":{"observed_at":"2026-08-16T11:57:01.868044Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:57:01.871664Z","title":"Judging llm-as- a-judge with mt-bench and chatbot arena","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.871664Z"},"links":{"citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:7f637c8c7bb43d075d3a9ef41d4d79808599ed13f192e5e3d34ae89fc3c4fc5c","observation_id":"be2da8b7-7aae-430a-80f0-7c12da66791f","resolution":{"observed_at":"2026-08-16T11:57:01.871664Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:57:01.874842Z","title":"Efficiently programming large language models using sglang","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.874842Z"},"links":{"citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:ea3beb5599c47c2fe3318725f57e4a27a4afa803fe97edc66658d52386add816","observation_id":"4000ee5f-5a5a-4768-85d1-4e767c690e33","resolution":{"observed_at":"2026-08-16T11:57:01.874842Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.09670","last_updated":"2024-06-06T15:50:51Z","snapshot_observed_at":"2026-08-16T14:26:37.115667Z","submitted_at":"2024-01-18T01:03:38Z","title":"DistServe: Disaggregating Prefill and Decoding for Goodput-optimized Large Language Model Serving","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.09670","snapshot_observed_at":"2026-08-16T11:57:01.877980Z","title":"Dist- serve: Disaggregating prefill and decoding for goodput- optimized large language model serving","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.877980Z"},"links":{"cited_paper":"/paper/2401.09670","citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:f0c719d4eff43adb5325a9af6c917b452814f1b78feaba407173e749fcbf93e3","observation_id":"eab5219c-9d8c-4cfc-97d3-ab0f472ae1f9","resolution":{"observed_at":"2026-08-16T11:57:01.877980Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.04675","last_updated":"2024-06-14T11:40:52Z","snapshot_observed_at":"2026-08-19T12:53:46.617614Z","submitted_at":"2023-04-10T15:51:30Z","title":"Multilingual Machine Translation with Large Language Models: Empirical Results and Analysis","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.04675","snapshot_observed_at":"2026-08-16T11:57:01.881688Z","title":"Multilingual machine translation with large language models: Empirical results and analysis","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-16T11:57:01.881688Z"},"links":{"cited_paper":"/paper/2304.04675","citing_paper":"/paper/2505.03756"},"observation_digest":"sha256:574d0bd008ab152df3e2fb8e96cb72b45c685e59c7b8984dc681d5fce81a616f","observation_id":"7075f73d-61e6-4f70-85b1-d89fbb0c83ae","resolution":{"observed_at":"2026-08-16T11:57:01.881688Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.03756","last_updated":"2025-04-19T13:17:34Z","latest_version":1,"primary_category":"cs.AR","snapshot_observed_at":"2026-08-17T22:14:56.744677Z","submitted_at":"2025-04-19T13:17:34Z","title":"Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management"},"reference_resolution":{"displayed":54,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":1,"unresolved":34,"verified_exact":2,"verified_fuzzy":17},"total_outbound_references":54},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 19 August 2026, this Paper Citation Record lists 54 of 54 outbound references and 6 inbound Pith citation observations for arXiv:2505.03756."}