{"as_of":"2026-08-13T23:24:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:993e7493654eaf2165a6364685aac696e6a72e66f15201a206214f688e0e2842","coverage":[{"denominator":49,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":49,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T13:56:39.808556Z","state":"measured"},{"denominator":50,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":50,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-12T01:08:14.247342Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-12T08:26:24.739905Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"cited_work":{"arxiv_id":"2412.12639","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.12639","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Falcon: Faster and parallel inference of large language models through enhanced semi-autoregressive drafting and custom-designed decoding tree","venue":null,"work_id":"66667291-fcf4-46da-a9da-821b1f13753d","year":2024},"citing_paper":{"arxiv_id":"2605.08632","last_updated":"2026-05-09T02:50:58Z","snapshot_observed_at":"2026-08-12T22:50:31.398809Z","submitted_at":"2026-05-09T02:50:58Z","title":"PARD-2: Target-Aligned Parallel Draft Model for Dual-Mode Speculative Decoding","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-12T01:08:14.247342Z"},"links":{"cited_paper":"/paper/2412.12639","citing_paper":"/paper/2605.08632"},"observation_digest":"sha256:19749afca2c47a370cc5a64f55c8e42e71ec48d67853e0854c340c99c4e3f00f","observation_id":"4e8c7bb5-17af-47b6-a400-6099c7817c98","resolution":{"observed_at":"2026-05-12T08:26:24.741992Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2412.12639/citation-record","integrity":"/paper/2412.12639/integrity","json":"/paper/2412.12639/citation-record.json","paper":"/paper/2412.12639"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:39.544245Z","title":", \" * write output.state after.block = add.period write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.544245Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:0b716da74485977e41f3ebfae690478f8e62b4a2378db46686273e71f3975dd8","observation_id":"088995a7-78ae-4a33-96d8-086b3ba695a1","resolution":{"observed_at":"2026-08-11T13:56:39.544245Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:39.549623Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.549623Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:096496627d9c2ab12450ba3c264d7f45c6f4672fe997adeb7325b267e5c647c4","observation_id":"1095de3a-02e5-48b5-8e39-cdeb02d13c4d","resolution":{"observed_at":"2026-08-11T13:56:39.549623Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.556071Z","title":null,"venue":null,"work_id":"beeac847-b0e6-4632-a5c2-8743561cd189","year":2022},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.555810Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:3ee0dcad046ca6e4532bc483936857426b205cd82bf52f3ac75def88cf3ca83d","observation_id":"cd21b300-3a5a-4f28-bccd-710ef8584d9c","resolution":{"observed_at":"2026-08-11T13:56:40.560090Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.10774","last_updated":"2024-06-14T23:32:32Z","snapshot_observed_at":"2026-07-06T17:17:56.276857Z","submitted_at":"2024-01-19T15:48:40Z","title":"Medusa: Simple LLM Inference Acceleration Framework with Multiple Decoding Heads","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.10774","snapshot_observed_at":"2026-08-11T13:56:39.561574Z","title":"D.; Chen, D.; and Dao, T","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.561574Z"},"links":{"cited_paper":"/paper/2401.10774","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:a646b1c1fa780890b61563815452068ce57d08543e59dc5a1801a966dc35585c","observation_id":"1f4542ba-b36b-4454-b281-de72137be92f","resolution":{"observed_at":"2026-08-11T13:56:39.561574Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.01318","last_updated":"2023-02-02T18:44:11Z","snapshot_observed_at":"2026-08-13T23:21:49.907837Z","submitted_at":"2023-02-02T18:44:11Z","title":"Accelerating Large Language Model Decoding with Speculative Sampling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.01318","snapshot_observed_at":"2026-08-11T13:56:39.568070Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.568070Z"},"links":{"cited_paper":"/paper/2302.01318","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:13844734d5b0091d81bb1f97119732f6a1ac96c48b213820bf073fb24e0452fb","observation_id":"c8fa522f-0984-42ad-99c2-8f573c815683","resolution":{"observed_at":"2026-08-11T13:56:39.568070Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2107.03374","last_updated":"2021-07-14T17:16:02Z","snapshot_observed_at":"2026-08-08T11:58:24.516369Z","submitted_at":"2021-07-07T17:41:24Z","title":"Evaluating Large Language Models Trained on Code","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2107.03374","snapshot_observed_at":"2026-08-11T13:56:39.573168Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.573168Z"},"links":{"cited_paper":"/paper/2107.03374","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:3f4f8522ef6648b2100d577703f4b8e6481c777605847b5a0511e4e884553885","observation_id":"b102a94c-ac3e-486c-b195-872c3f3f1a09","resolution":{"observed_at":"2026-08-11T13:56:39.573168Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11462","last_updated":"2025-07-13T19:45:40Z","snapshot_observed_at":"2026-08-13T23:01:29.703807Z","submitted_at":"2023-12-18T18:59:46Z","title":"Cascade Speculative Drafting for Even Faster LLM Inference","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11462","snapshot_observed_at":"2026-08-11T13:56:39.580987Z","title":"C.-C.; and Huang, J","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.580987Z"},"links":{"cited_paper":"/paper/2312.11462","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:4d5e23efc925a5ecb70b96634dfef02dda156d507450f6b03d6c3a733344051b","observation_id":"85d3b22d-8c13-4fad-9bc1-0a8201207dca","resolution":{"observed_at":"2026-08-11T13:56:39.580987Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-11T13:56:39.586758Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.586758Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:a1d747bbd35c1ce7728d01f5b6f1a66dff8b295d64202353cfe9d8cc78df91a4","observation_id":"49babc3b-b03a-4f88-9090-b3237ccf8188","resolution":{"observed_at":"2026-08-11T13:56:39.586758Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.542624Z","title":"F.; Tao, D.; and Tu, Z","venue":null,"work_id":"31959b18-bbc0-4b2c-95b5-0d2fc9997b67","year":2021},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.592117Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:2cf914606ef2b4d06687fa88e07a0f6d0f0f19b39f79a65375ba9b7aed427c03","observation_id":"330b6d1d-7df0-4e21-8124-61eeafdf9cd6","resolution":{"observed_at":"2026-08-11T13:56:40.547022Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.10360","last_updated":"2022-03-17T11:49:55Z","snapshot_observed_at":"2026-07-06T10:51:12.871009Z","submitted_at":"2021-03-18T16:30:26Z","title":"GLM: General Language Model Pretraining with Autoregressive Blank Infilling","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.10360","snapshot_observed_at":"2026-08-11T13:56:39.596093Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.596093Z"},"links":{"cited_paper":"/paper/2103.10360","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:aa88482f1730bea6e8801a29af5e3c4de6dd007541d337f972cdd51c9c716ed1","observation_id":"dfc0bfd4-2231-4a0d-b1f1-898a3cd10303","resolution":{"observed_at":"2026-08-11T13:56:39.596093Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.530666Z","title":null,"venue":null,"work_id":"5c2f64cf-bc89-4343-b868-c1317585f30a","year":2021},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.600341Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:c1936b58168cadb8603aa75381d525bedb5808b35bffc24b039d137b893240e2","observation_id":"2ada1f50-1235-4c8d-a9bc-f04c24a8fe48","resolution":{"observed_at":"2026-08-11T13:56:40.534753Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.515082Z","title":null,"venue":null,"work_id":"4a87edca-6fa3-489e-8f84-2ff6ee31a37b","year":2019},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.603874Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:7a2a8e49936a6bbff5261ac1b263becc2508039815b96bc3445ab58f3d49b484","observation_id":"6f84257a-97a4-414e-9fef-66e472a28cbf","resolution":{"observed_at":"2026-08-11T13:56:40.519615Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.19737","last_updated":"2024-04-30T17:33:57Z","snapshot_observed_at":"2026-08-12T13:06:04.804423Z","submitted_at":"2024-04-30T17:33:57Z","title":"Better & Faster Large Language Models via Multi-token Prediction","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.19737","snapshot_observed_at":"2026-08-11T13:56:39.607525Z","title":"Y.; Rozière, B.; Lopez-Paz, D.; and Synnaeve, G","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.607525Z"},"links":{"cited_paper":"/paper/2404.19737","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:806695969ae47d07595f750adcbb92325624ea191dc45107e7adeb3115f2c85a","observation_id":"175d18a3-778f-45f0-a3f4-316ff51a0d06","resolution":{"observed_at":"2026-08-11T13:56:39.607525Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:39.612401Z","title":null,"venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.612401Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:aa8f5ef44d4ca492df67fb766051098aa636726104ba2b7388748cbb6d5a89dc","observation_id":"af66018e-a323-43ff-a041-e8aa6ef34276","resolution":{"observed_at":"2026-08-11T13:56:39.612401Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.489084Z","title":null,"venue":null,"work_id":"277bb584-0ea1-40a5-8e8d-16652e72540b","year":2019},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.616664Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:8e8d81659ec5ce26db93aa5a3efcc0f39a5dfcd1d7c9e688460c9b38e66095ab","observation_id":"8ab821b5-56ec-4d84-af51-00a24bbfdb33","resolution":{"observed_at":"2026-08-11T13:56:40.493299Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1911.08717","last_updated":"2019-11-21T09:43:45Z","snapshot_observed_at":"2026-08-03T00:59:52.298231Z","submitted_at":"2019-11-20T05:48:31Z","title":"Fine-Tuning by Curriculum Learning for Non-Autoregressive Neural Machine Translation","version":2},"cited_work":{"arxiv_id":"1911.08717","doi":null,"metadata_source":"pith","pith_arxiv_id":"1911.08717","snapshot_observed_at":"2026-08-11T13:56:40.080049Z","title":"Fine-Tuning by Curriculum Learning for Non-Autoregressive Neural Machine Translation","venue":"cs.LG","work_id":"2ca4dab0-c145-48d7-9e98-5fa577606994","year":2019},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.621807Z"},"links":{"cited_paper":"/paper/1911.08717","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:a140047fa5268ee7777fe0633e2ef94f55893165b0b27fac9efff061f6201402","observation_id":"8fa911fd-5676-4554-b832-7d0412bbc4be","resolution":{"observed_at":"2026-08-11T13:56:40.085206Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.472194Z","title":null,"venue":null,"work_id":"22100308-796c-4670-9e53-8426ea458e3b","year":2020},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.627068Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:69da80fca0af1bd77602a93e3c62ec8ae46a50a435f474253fa1ead2efc33ba6","observation_id":"a288e5bd-b530-420d-939f-c51e55f907c5","resolution":{"observed_at":"2026-08-11T13:56:40.477145Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.11640","last_updated":"2021-12-22T03:06:27Z","snapshot_observed_at":"2026-08-13T17:10:23.806896Z","submitted_at":"2021-12-22T03:06:27Z","title":"Self-Distillation Mixup Training for Non-autoregressive Neural Machine Translation","version":1},"cited_work":{"arxiv_id":"2112.11640","doi":null,"metadata_source":"pith","pith_arxiv_id":"2112.11640","snapshot_observed_at":"2026-08-11T13:56:40.054637Z","title":"Self-Distillation Mixup Training for Non-autoregressive Neural Machine Translation","venue":"cs.CL","work_id":"44224712-0055-430e-9212-0aa052d1d177","year":2021},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.631862Z"},"links":{"cited_paper":"/paper/2112.11640","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:55dc74f38cdfe98773402155d4b90f3e4635ec9a2069ea9a95d91f29ceaf8d39","observation_id":"4482b961-7191-436d-a1f9-4a51fd5f0c59","resolution":{"observed_at":"2026-08-11T13:56:40.061510Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1503.02531","last_updated":"2015-03-09T15:44:49Z","snapshot_observed_at":"2026-07-06T04:11:24.157003Z","submitted_at":"2015-03-09T15:44:49Z","title":"Distilling the Knowledge in a Neural Network","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1503.02531","snapshot_observed_at":"2026-08-11T13:56:39.638681Z","title":null,"venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.638681Z"},"links":{"cited_paper":"/paper/1503.02531","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:d7f884db53fe8f73f771bf40ac67b93914dee6e34b448247a4842f91d260bbea","observation_id":"e903e635-f4ba-40a0-a7b8-19bfd2aff34f","resolution":{"observed_at":"2026-08-11T13:56:39.638681Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:39.642790Z","title":null,"venue":null,"work_id":null,"year":1997},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.642790Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:2d196af1b2ad0195ca417434fac7664831a7e7e24ac576f5625c79cae46890d8","observation_id":"47ab6896-f1db-4b66-b3b6-ac9762015293","resolution":{"observed_at":"2026-08-11T13:56:39.642790Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.439155Z","title":"W.; Gholami, A.; and Keutzer, K","venue":null,"work_id":"0e439127-7e01-48e4-8675-3f5547b09844","year":2023},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.647897Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:445c70790bd77584d87cbb3528dc92f543cd6b71d01a5d64234a326b01fb1cb3","observation_id":"176e4f2f-8909-4647-a87e-36107980e473","resolution":{"observed_at":"2026-08-11T13:56:40.446329Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2211.17192","last_updated":"2023-05-18T20:28:20Z","snapshot_observed_at":"2026-08-08T11:03:33.403601Z","submitted_at":"2022-11-30T17:33:28Z","title":"Fast Inference from Transformers via Speculative Decoding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.17192","snapshot_observed_at":"2026-08-11T13:56:39.653053Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.653053Z"},"links":{"cited_paper":"/paper/2211.17192","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:360a6f9cb0cd2c3b678571f9e7317e43ca0198f0d8d3d63ce8b6faf874ff50d6","observation_id":"ba1f2e9a-4969-45f3-9bfd-3d67295ac2cc","resolution":{"observed_at":"2026-08-11T13:56:39.653053Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.15077","last_updated":"2025-03-04T13:58:39Z","snapshot_observed_at":"2026-08-03T09:40:31.365295Z","submitted_at":"2024-01-26T18:59:01Z","title":"EAGLE: Speculative Sampling Requires Rethinking Feature Uncertainty","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.15077","snapshot_observed_at":"2026-08-11T13:56:39.663532Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.663532Z"},"links":{"cited_paper":"/paper/2401.15077","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:c7933a9e4bcf6c31d36c12503afde720db2efdf308b296aab168fc08c4fbcb91","observation_id":"018023f6-a511-40a7-abf7-378e706924c6","resolution":{"observed_at":"2026-08-11T13:56:39.663532Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.421550Z","title":null,"venue":null,"work_id":"bc3bdbce-b3d5-4cd8-a0d8-af64bba1dd84","year":2024},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.670302Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:50e09ba17e1f21824ab36cf39c2d0da99f60f6a83cc4735d37d4f321c73ebd58","observation_id":"fd32a44b-b5b5-45b3-80be-770b28e89738","resolution":{"observed_at":"2026-08-11T13:56:40.426314Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.13581","last_updated":"2023-11-22T18:37:27Z","snapshot_observed_at":"2026-08-13T05:19:28.412612Z","submitted_at":"2023-11-22T18:37:27Z","title":"PaSS: Parallel Speculative Sampling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.13581","snapshot_observed_at":"2026-08-11T13:56:39.674571Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.674571Z"},"links":{"cited_paper":"/paper/2311.13581","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:59b65e18b2aba6d63d9a4fc3bd59c0d7a25d3485d759f574a433b407df99ad1f","observation_id":"9c9e0826-8656-4623-89a3-6b8b1145ee52","resolution":{"observed_at":"2026-08-11T13:56:39.674571Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2209.11895","last_updated":"2022-09-24T00:43:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-09-24T00:43:19Z","title":"In-context Learning and Induction Heads","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.11895","snapshot_observed_at":"2026-08-11T13:56:39.679327Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.679327Z"},"links":{"cited_paper":"/paper/2209.11895","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:2def9e3f027883575d1f5ceb0a2ba767f8bd0d0ed2548ceaa55bd34de4610521","observation_id":"dc02a5c5-cb1a-44f5-b382-3ce5f730c8e1","resolution":{"observed_at":"2026-08-11T13:56:39.679327Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.404189Z","title":null,"venue":null,"work_id":"0a99a7b9-bb9d-4069-a638-b3967b0614cb","year":2021},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.688977Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:473868451d7458cc920f4727660b80b8ec045f08b6a0788eee8a0c28d5e786c0","observation_id":"85628b68-1705-4171-bacb-ad0d145d294f","resolution":{"observed_at":"2026-08-11T13:56:40.408816Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.383741Z","title":null,"venue":null,"work_id":"32d774b7-fdce-4040-99b5-edfa9a9b9b4e","year":2020},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.693837Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:360d0ab27cc8ed41644025bf0cd384464efeb74dc5a29450498e9163e8c5e9dc","observation_id":"9885abe7-7969-4c1a-8386-d0338e15da0a","resolution":{"observed_at":"2026-08-11T13:56:40.388929Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.361580Z","title":null,"venue":null,"work_id":"585816b2-626c-40ce-9b55-4f40f7080e42","year":1992},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.699417Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:47b5e234178135a7b975ff317567ac9d31bc170a86742252461336f13289c1d0","observation_id":"d0236f5d-c9bd-4bdf-8d7c-04b16e9ac3d0","resolution":{"observed_at":"2026-08-11T13:56:40.369184Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.342621Z","title":null,"venue":null,"work_id":"e933b458-94d9-4b90-b16e-708d6476dc62","year":2023},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.707155Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:69796fc5961bdb5f02b81dfe4789d928416740ff0c34178ae040d5bc18bda2e9","observation_id":"7ff372d0-6ce5-423d-9cf6-8092dfad27ca","resolution":{"observed_at":"2026-08-11T13:56:40.348487Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.324082Z","title":null,"venue":null,"work_id":"a3f390f7-1ef5-42b2-a20b-ed8fdbfad54a","year":2023},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.713796Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:942c022fbe00ef3047a72a76e6ad4e49a92e0886323c82eb486668311bb1eaa7","observation_id":"dd547bb5-cfd0-4fcf-9315-2da5ce487123","resolution":{"observed_at":"2026-08-11T13:56:40.329362Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.04623","last_updated":"2023-08-08T23:29:55Z","snapshot_observed_at":"2026-08-13T10:38:17.259933Z","submitted_at":"2023-08-08T23:29:55Z","title":"Accelerating LLM Inference with Staged Speculative Decoding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.04623","snapshot_observed_at":"2026-08-11T13:56:39.717832Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.717832Z"},"links":{"cited_paper":"/paper/2308.04623","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:c17c7f5b27a4de56e7499107893c3ff5c6af3af5a259a2071d9b2f0db94949cb","observation_id":"975475f5-45a8-4cae-b9be-ff813764f66e","resolution":{"observed_at":"2026-08-11T13:56:39.717832Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.309008Z","title":null,"venue":null,"work_id":"955d4649-7c03-44f6-8260-8874a3d62258","year":2018},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.723725Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:2c3d69369a2a6715927acc79323d80b4022305469711faf46308e416f566fe57","observation_id":"209fa9cc-c15f-45fa-a7bc-36baa9148d3a","resolution":{"observed_at":"2026-08-11T13:56:40.313913Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.292599Z","title":null,"venue":null,"work_id":"04defcc9-002b-4e82-90ac-524e8b91dad2","year":2019},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.729611Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:515cf63fd8a43fe7aa977176a746019520ee30781c255f8abc2c95c6e3bb36c4","observation_id":"62d5b023-e51c-4dfe-a105-c319e0e4f94e","resolution":{"observed_at":"2026-08-11T13:56:40.298005Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1609.03499","last_updated":"2016-09-19T18:04:35Z","snapshot_observed_at":"2026-08-13T01:24:25.628327Z","submitted_at":"2016-09-12T17:29:40Z","title":"WaveNet: A Generative Model for Raw Audio","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1609.03499","snapshot_observed_at":"2026-08-11T13:56:39.734235Z","title":null,"venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.734235Z"},"links":{"cited_paper":"/paper/1609.03499","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:fd387d5de74022c2fc014e2641223e3586f974e44f3827544f02f6136d4d82d4","observation_id":"15dee101-6225-40c4-b916-5ceafb872b5e","resolution":{"observed_at":"2026-08-11T13:56:39.734235Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.275240Z","title":null,"venue":null,"work_id":"de26a3bf-d67b-4633-a302-e2602091d7b6","year":2018},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.739625Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:3c8c8e0afb18eac1d4dc9ff2f58587c8cec2cc9cbb97a73b6a23438690469c19","observation_id":"329f5adc-5855-4742-a60b-bdb427d69611","resolution":{"observed_at":"2026-08-11T13:56:40.280552Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.03863","last_updated":"2024-05-23T06:08:37Z","snapshot_observed_at":"2026-08-13T05:08:25.522951Z","submitted_at":"2023-12-06T19:18:42Z","title":"Efficient Large Language Models: A Survey","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.03863","snapshot_observed_at":"2026-08-11T13:56:39.744899Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.744899Z"},"links":{"cited_paper":"/paper/2312.03863","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:c4f6b71800afe4e6fd44b3a683a34db404cec554655c3eab4c487b20ff0c01ce","observation_id":"162cc3db-c0d0-45ed-b514-07c4426e776d","resolution":{"observed_at":"2026-08-11T13:56:39.744899Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.255618Z","title":null,"venue":null,"work_id":"d460a62b-eafe-4106-929c-3e996bc08021","year":2018},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.749541Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:153b7f2c6c7ebf73664d27aa9c62ce30e43a931f16eea1c7d74615a1d5c8efa9","observation_id":"9f89cce7-c135-44d4-aa75-2e5694ae6d62","resolution":{"observed_at":"2026-08-11T13:56:40.260481Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.237391Z","title":null,"venue":null,"work_id":"3e4cfe43-cfeb-4ef4-83dd-13feb6e49f41","year":2019},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.756167Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:5c105fa7b1942dd34d9011ff49b7f418e917c6b5678cf5a5653f37be1bbf6fc8","observation_id":"560487e9-aabd-4d9c-9d73-6e1b71cf50c8","resolution":{"observed_at":"2026-08-11T13:56:40.242349Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.19124","last_updated":"2024-06-06T18:38:34Z","snapshot_observed_at":"2026-08-13T00:19:04.418458Z","submitted_at":"2024-04-29T21:59:07Z","title":"Accelerating Production LLMs with Combined Token/Embedding Speculators","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.19124","snapshot_observed_at":"2026-08-11T13:56:39.760297Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.760297Z"},"links":{"cited_paper":"/paper/2404.19124","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:40e8b8c8a52a78fd888b76594c6a6ddc3580361a7461e51629e22f8393b50e2c","observation_id":"399bc475-9fcd-49fc-b503-f18391603f7e","resolution":{"observed_at":"2026-08-11T13:56:39.760297Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.213267Z","title":null,"venue":null,"work_id":"e2522118-c974-4af1-803c-f84fceba3955","year":2023},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.765552Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:5fe31d38e840051098df6644ed3eb864bdc1975e059fe77e86570ba245f8702e","observation_id":"9e19fcaf-326a-4171-89ea-ff63a7fc60b1","resolution":{"observed_at":"2026-08-11T13:56:40.217318Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-08-13T04:42:35.832870Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-08-11T13:56:39.772067Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.772067Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:6b465005151cf0df30b2f41f425259ee02c2a1590fac52420d2b236e84361c3b","observation_id":"6a36a5c1-318b-4044-b772-3322b34f97d5","resolution":{"observed_at":"2026-08-11T13:56:39.772067Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2204.09269","last_updated":"2023-07-06T07:29:23Z","snapshot_observed_at":"2026-08-13T15:59:12.729827Z","submitted_at":"2022-04-20T07:25:22Z","title":"A Survey on Non-Autoregressive Generation for Neural Machine Translation and Beyond","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.09269","snapshot_observed_at":"2026-08-11T13:56:39.777354Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.777354Z"},"links":{"cited_paper":"/paper/2204.09269","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:7841a7ae0d9367d7cdcd329ac8aa0ab42c8af262d1c056f5ef8ffb8dea75359e","observation_id":"355893e3-66d0-4fdb-9c4b-b7356dafd000","resolution":{"observed_at":"2026-08-11T13:56:39.777354Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.198755Z","title":null,"venue":null,"work_id":"14952ba5-f1b9-4ca4-b4d0-865a1e61ca7d","year":2021},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.782758Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:5c60b3eed8c52c75b32ef0d6d38c3bc7be1b3100dbe6d41485f15e98583cc960","observation_id":"eecc80ce-032a-4d7f-a8c0-6b2e2fff821b","resolution":{"observed_at":"2026-08-11T13:56:40.202894Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.08168","last_updated":"2024-05-20T02:37:20Z","snapshot_observed_at":"2026-08-13T10:13:11.986500Z","submitted_at":"2023-09-15T05:34:32Z","title":"Draft & Verify: Lossless Large Language Model Acceleration via Self-Speculative Decoding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.08168","snapshot_observed_at":"2026-08-11T13:56:39.788351Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.788351Z"},"links":{"cited_paper":"/paper/2309.08168","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:6acb36c138b12586a3ee67c1b566e5753fc052dac4f3bdb6081eef12bb5ab7d2","observation_id":"368bf258-9b1c-4e01-a035-843ea2b7ba09","resolution":{"observed_at":"2026-08-11T13:56:39.788351Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.12728","last_updated":"2024-05-30T11:25:08Z","snapshot_observed_at":"2026-08-13T04:58:00.331388Z","submitted_at":"2023-12-20T02:55:15Z","title":"Lookahead: An Inference Acceleration Framework for Large Language Model with Lossless Generation Accuracy","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.12728","snapshot_observed_at":"2026-08-11T13:56:39.793794Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.793794Z"},"links":{"cited_paper":"/paper/2312.12728","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:d51f890be58ad6568fc9e5600511c25b9621d89e7a559fa0a27a60f8effd129f","observation_id":"871eb455-5dcf-4a0b-8ac6-99956511008e","resolution":{"observed_at":"2026-08-11T13:56:39.793794Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:39.798834Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.798834Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:c2ae0488ed023c23cf43c5be5d4459b93582e63cda4894ee346cbce18463443a","observation_id":"1320dda8-e617-4b36-adfa-b0a7a0f8b86a","resolution":{"observed_at":"2026-08-11T13:56:39.798834Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.07633","last_updated":"2024-07-30T13:14:55Z","snapshot_observed_at":"2026-08-13T10:34:25.802574Z","submitted_at":"2023-08-15T08:31:05Z","title":"A Survey on Model Compression for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.07633","snapshot_observed_at":"2026-08-11T13:56:39.803614Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.803614Z"},"links":{"cited_paper":"/paper/2308.07633","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:57c1598dd295625d414f228be9be4b3c57ccfc27e4088873a09b5aa257578236","observation_id":"b884096a-6e81-40bd-ac00-1248f41f1541","resolution":{"observed_at":"2026-08-11T13:56:39.803614Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1803.03382","last_updated":"2018-06-07T21:48:19Z","snapshot_observed_at":"2026-08-08T23:17:33.712950Z","submitted_at":"2018-03-09T04:39:35Z","title":"Fast Decoding in Sequence Models using Discrete Latent Variables","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1803.03382","snapshot_observed_at":"2026-08-11T13:56:39.808556Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.808556Z"},"links":{"cited_paper":"/paper/1803.03382","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:36adcbe27621844d8b152929d58e478afaa8df12c9cd6ca9c2121f3c3395971a","observation_id":"6908f2f6-4ebb-4d34-8076-e355979e08aa","resolution":{"observed_at":"2026-08-11T13:56:39.808556Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","latest_version":3,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-13T12:29:02.693937Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree"},"reference_resolution":{"displayed":49,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":45,"verified_exact":2,"verified_fuzzy":2},"total_outbound_references":49},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 49 of 49 outbound references and 1 inbound Pith citation observation for arXiv:2412.12639."}