{"as_of":"2026-08-14T00:35:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:2f11c38af234b356aa8e214e4e47d9fe0539884de80b4d62f87de7ea347167ad","coverage":[{"denominator":50,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":50,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T22:19:03.875806Z","state":"measured"},{"denominator":52,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":52,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-27T01:05:20.432379Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T20:58:57.478873Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"cited_work":{"arxiv_id":"2506.22023","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.22023","snapshot_observed_at":"2026-07-03T20:58:57.478873Z","title":"InProceedings of the 40th Interna- tional Conference on Machine Learning (ICML)","venue":null,"work_id":"24957ecd-3b05-49da-83db-db9bc22b59a9","year":2025},"citing_paper":{"arxiv_id":"2603.04592","last_updated":"2026-04-19T16:23:58Z","snapshot_observed_at":"2026-08-03T02:05:46.790260Z","submitted_at":"2026-03-04T20:37:30Z","title":"From Static Inference to Dynamic Interaction: A Survey of Streaming Large Language Models","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-15T16:16:15.819622Z"},"links":{"cited_paper":"/paper/2506.22023","citing_paper":"/paper/2603.04592"},"observation_digest":"sha256:07b7a4144219225622a25e633cb48089660fb489bfd95912cf2a3fe3e6344234","observation_id":"03bae24a-489b-4704-8727-8f4f7b6df872","resolution":{"observed_at":"2026-05-15T16:20:09.991805Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"cited_work":{"arxiv_id":"2506.22023","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.22023","snapshot_observed_at":"2026-07-03T20:58:57.478873Z","title":"InProceedings of the 40th Interna- tional Conference on Machine Learning (ICML)","venue":null,"work_id":"24957ecd-3b05-49da-83db-db9bc22b59a9","year":2025},"citing_paper":{"arxiv_id":"2606.18319","last_updated":"2026-06-22T12:15:05Z","snapshot_observed_at":"2026-08-06T15:28:39.341114Z","submitted_at":"2026-06-16T13:43:14Z","title":"ASTRA: A Scalable Next-Generation ATCO Training Simulator with Autonomous Simpilots","version":2},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-06-27T01:05:20.432379Z"},"links":{"cited_paper":"/paper/2506.22023","citing_paper":"/paper/2606.18319"},"observation_digest":"sha256:5b71fa75dacdb5c9f4d5cd0ed4a1aa4275da93e5fcaa067fcb2a5c78a589e484","observation_id":"e5d064c5-efdc-4ec2-aab7-be973da6cf34","resolution":{"observed_at":"2026-07-03T20:58:57.480674Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.22023/citation-record","integrity":"/paper/2506.22023/integrity","json":"/paper/2506.22023/citation-record.json","paper":"/paper/2506.22023"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.18223","last_updated":"2026-03-18T05:34:39Z","snapshot_observed_at":"2026-08-06T23:27:24.356320Z","submitted_at":"2023-03-31T17:28:46Z","title":"A Survey of Large Language Models","version":19},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.18223","snapshot_observed_at":"2026-08-06T22:18:58.451990Z","title":"A survey of large language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T22:18:58.451990Z"},"links":{"cited_paper":"/paper/2303.18223","citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:5721e5de6ab2711d99d3a1ea9d56f110aa918d3202a0e005c5662784b13789ab","observation_id":"a5e68496-bb31-450b-84ca-0952e784f0c2","resolution":{"observed_at":"2026-08-06T22:18:58.451990Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.13071","last_updated":"2024-09-18T12:02:47Z","snapshot_observed_at":"2026-08-13T04:14:27.294775Z","submitted_at":"2024-02-20T15:13:38Z","title":"Codec-SUPERB: An In-Depth Analysis of Sound Codec Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.13071","snapshot_observed_at":"2026-08-06T22:18:58.537491Z","title":"Codec-SUPERB: An In-Depth Analysis of Sound Codec Models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T22:18:58.537491Z"},"links":{"cited_paper":"/paper/2402.13071","citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:d87694c136adf933ed00f18bf3c00efe4a3856c46d3eb6e6e7f60e251adea08b","observation_id":"cb1ccaf6-217f-4166-bf0c-6b91003ee642","resolution":{"observed_at":"2026-08-06T22:18:58.537491Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:18:58.693492Z","title":"Recent advances in discrete speech tokens: A review,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T22:18:58.693492Z"},"links":{"citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:25f964049c23cd3972e5914506de9f8a1fb15c80ad4c4bb4d5d2c5a7b9db26e3","observation_id":"bd8c8cb0-03ac-476c-8e5b-b72ad47d999e","resolution":{"observed_at":"2026-08-06T22:18:58.693492Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.08528","last_updated":"2026-04-07T06:11:20Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-11T13:40:53Z","title":"On The Landscape of Spoken Language Models: A Comprehensive Survey","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.08528","snapshot_observed_at":"2026-08-06T22:18:58.895945Z","title":"On the landscape of spoken language models: A comprehensive survey,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T22:18:58.895945Z"},"links":{"cited_paper":"/paper/2504.08528","citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:e935dda45059f6790fedb5ee382cfddd2de4a224129740f5251bf7dff5820002","observation_id":"8ef43034-e09c-47b0-b03f-f0e7494a5942","resolution":{"observed_at":"2026-08-06T22:18:58.895945Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:19:09.227450Z","title":"Neural Discrete Representation Learning,","venue":null,"work_id":"fb6fddb6-c7a5-41e6-bbd0-6a38d560a00a","year":2017},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T22:18:59.021254Z"},"links":{"citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:9479078227c95a050943477d0681b1b92d6ea0ff846ff58c5ebf835c5a181819","observation_id":"170c9bb8-d83e-4618-bde3-c75ad123e85e","resolution":{"observed_at":"2026-08-06T22:19:09.288020Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:19:09.002266Z","title":"Fast and high-quality auto-regressive speech synthesis via speculative decoding,","venue":null,"work_id":"d6ab3250-a352-4e85-97a1-f5be42d49329","year":2025},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T22:18:59.166926Z"},"links":{"citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:002cfd9e496ed7de9d8edee38830eb55087582901e21c75c5f315ee81f40d66e","observation_id":"7d465b53-a010-46e1-83f7-f162dcbfc1dd","resolution":{"observed_at":"2026-08-06T22:19:09.091961Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.13839","last_updated":"2024-10-17T17:55:26Z","snapshot_observed_at":"2026-08-12T22:20:39.929273Z","submitted_at":"2024-10-17T17:55:26Z","title":"Accelerating Codec-based Speech Synthesis with Multi-Token Prediction and Speculative Decoding","version":1},"cited_work":{"arxiv_id":"2410.13839","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.13839","snapshot_observed_at":"2026-08-06T22:19:04.621354Z","title":"Accelerating Codec-based Speech Synthesis with Multi-Token Prediction and Speculative Decoding","venue":"cs.SD","work_id":"f5e1dc75-a8c7-4a12-8d3f-64dcc6bb262a","year":2024},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T22:18:59.328921Z"},"links":{"cited_paper":"/paper/2410.13839","citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:ddf8b1d0593e2a480b0e7355e8d43dfa443dc16f953cd4708606777d16adedaa","observation_id":"e7cfcf5d-e68c-4595-8ce2-625a9860e24b","resolution":{"observed_at":"2026-08-06T22:19:04.696229Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.04060","last_updated":"2025-04-22T07:59:31Z","snapshot_observed_at":"2026-08-13T04:13:36.575098Z","submitted_at":"2025-04-05T04:57:12Z","title":"VocalNet: Speech LLM with Multi-Token Prediction for Faster and High-Quality Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.04060","snapshot_observed_at":"2026-08-06T22:18:59.507675Z","title":"V ocalnet: Speech llm with multi-token prediction for faster and high-quality generation,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T22:18:59.507675Z"},"links":{"cited_paper":"/paper/2504.04060","citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:40a38df8208cf52118ab51af9894c9070c80a04a0fb96c7bfbb92b20a67dd2e9","observation_id":"7f734fbc-4e28-473c-b2b5-189bfff8215b","resolution":{"observed_at":"2026-08-06T22:18:59.507675Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-06T22:18:59.608837Z","title":"Deepseekmath: Pushing the limits of mathematical reasoning in open language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T22:18:59.608837Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:5484949e3ce3c15e02d86a05631f022dc02bd785cf414adf5199d100938ec6e8","observation_id":"4b95c2a2-7701-426e-9b56-1e2a926f9b7d","resolution":{"observed_at":"2026-08-06T22:18:59.608837Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17161","last_updated":"2025-05-26T17:16:45Z","snapshot_observed_at":"2026-08-09T18:26:12.869738Z","submitted_at":"2025-01-28T18:59:44Z","title":"SFT Memorizes, RL Generalizes: A Comparative Study of Foundation Model Post-training","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.17161","snapshot_observed_at":"2026-08-06T22:18:59.735703Z","title":"Sft memorizes, rl generalizes: A comparative study of foundation model post-training,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T22:18:59.735703Z"},"links":{"cited_paper":"/paper/2501.17161","citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:a19cfc2074623484938fd009109d342456ebcc30542935616dac917a90a3c45a","observation_id":"ac48b1a8-3cfd-458a-8666-e40d88a486a5","resolution":{"observed_at":"2026-08-06T22:18:59.735703Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.13837","last_updated":"2025-11-24T06:11:04Z","snapshot_observed_at":"2026-07-06T21:11:34.701779Z","submitted_at":"2025-04-18T17:59:56Z","title":"Does Reinforcement Learning Really Incentivize Reasoning Capacity in LLMs Beyond the Base Model?","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.13837","snapshot_observed_at":"2026-08-06T22:18:59.877417Z","title":"Does reinforcement learning really incentivize reasoning capacity in llms beyond the base model?","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T22:18:59.877417Z"},"links":{"cited_paper":"/paper/2504.13837","citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:e190adea67a99acc809dcea910dd83e1802abfa4b4e9dea57fd80c8b9bc35c79","observation_id":"0eaed673-25f5-44ed-9d4a-9084025b75a2","resolution":{"observed_at":"2026-08-06T22:18:59.877417Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:19:08.756059Z","title":"LibriTTS: A Corpus Derived from LibriSpeech for Text-to- Speech,","venue":null,"work_id":"5d247630-9fa7-408e-b7d7-eb3374458b10","year":2019},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:00.015587Z"},"links":{"citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:ffe20b396193c2735edebe96a843ec01426fa5a2993fe37545dfb5217121b813","observation_id":"10172a57-2cf3-4b8f-a1e9-81406f67e2dd","resolution":{"observed_at":"2026-08-06T22:19:08.859990Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:19:08.439451Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units,","venue":null,"work_id":"69e80720-5d69-42da-adae-978170c72556","year":2021},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:00.124151Z"},"links":{"citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:7d435061ffedd04142c6cbad06131fd3bf55d2d4aa4958dce6911398ce662157","observation_id":"b04355ef-2ce0-4264-8d3c-142da83fbb8a","resolution":{"observed_at":"2026-08-06T22:19:08.546115Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05407","last_updated":"2024-07-09T07:42:51Z","snapshot_observed_at":"2026-08-10T17:49:50.848957Z","submitted_at":"2024-07-07T15:16:19Z","title":"CosyVoice: A Scalable Multilingual Zero-shot Text-to-speech Synthesizer based on Supervised Semantic Tokens","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.05407","snapshot_observed_at":"2026-08-06T22:19:00.278755Z","title":"CosyV oice: A Scalable Multilingual Zero-Shot Text-to-Speech Synthesizer Based on Supervised Semantic Tokens,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:00.278755Z"},"links":{"cited_paper":"/paper/2407.05407","citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:a6d62ea77d0b6c3c111ccfee48685f580346a8b15344e85af79775af2490c0b4","observation_id":"f64dcb36-58b7-422b-83b9-42bbada1cba2","resolution":{"observed_at":"2026-08-06T22:19:00.278755Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:19:08.166715Z","title":"Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers,","venue":null,"work_id":"e7351c72-fea7-4be6-aa78-bfde0a40387c","year":2025},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:00.426911Z"},"links":{"citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:dba95d6227f47184b0c57b588340a126ca086756bcd7acedc7187958fe749325","observation_id":"cca8f652-0bf6-4cff-9abb-116298515508","resolution":{"observed_at":"2026-08-06T22:19:08.260615Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:19:07.846600Z","title":"V oiceCraft: Zero-Shot Speech Editing and Text-to-Speech in the Wild,","venue":null,"work_id":"4a32875c-7bb1-4f6b-af59-f67966ac3fc1","year":2024},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:00.547714Z"},"links":{"citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:d58c5a7d470fa0cb3f5ddb537457a87c7d3903de78031490fb64098904279b63","observation_id":"cbe6412f-31f3-4fdc-b33f-731c6b255e13","resolution":{"observed_at":"2026-08-06T22:19:07.967672Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.02430","last_updated":"2024-06-04T15:48:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-04T15:48:29Z","title":"Seed-TTS: A Family of High-Quality Versatile Speech Generation Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.02430","snapshot_observed_at":"2026-08-06T22:19:00.703156Z","title":"Seed-TTS: A Family of High-Quality Versatile Speech Generation Models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:00.703156Z"},"links":{"cited_paper":"/paper/2406.02430","citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:31184f6af5c16a4a27b706253c95a8755288795a41fce18767efa4fe7a20d44f","observation_id":"3fff0af9-2b2b-4938-aa93-41cff71acb69","resolution":{"observed_at":"2026-08-06T22:19:00.703156Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.08093","last_updated":"2024-02-15T18:57:26Z","snapshot_observed_at":"2026-08-13T04:20:12.197716Z","submitted_at":"2024-02-12T22:21:30Z","title":"BASE TTS: Lessons from building a billion-parameter Text-to-Speech model on 100K hours of data","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.08093","snapshot_observed_at":"2026-08-06T22:19:00.848288Z","title":"BASE TTS: Lessons from Building a Billion-Parameter Text-to-Speech Model on 100K Hours of Data,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:00.848288Z"},"links":{"cited_paper":"/paper/2402.08093","citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:da9d53c7539e263c7856bbb360e3779e0808e0ab657c880705a82aea7e2ece11","observation_id":"45eb768c-4d27-4724-8018-1a3d215099e3","resolution":{"observed_at":"2026-08-06T22:19:00.848288Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:19:07.607473Z","title":"ELLA-V: Stable Neural Codec Language Modelling with Alignment-Guided Sequence Reordering,","venue":null,"work_id":"3019cb95-b5e2-4c67-9da7-64f06873714a","year":2025},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:00.908080Z"},"links":{"citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:8a4c12525aa34b42a57e9ca3cf1bd2d82ac48ff0d5001428045d00fb3c815e99","observation_id":"0f544de7-5d73-42a3-aa1b-fde17836f6c3","resolution":{"observed_at":"2026-08-06T22:19:07.735392Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07855","last_updated":"2024-06-12T04:09:44Z","snapshot_observed_at":"2026-08-12T23:44:58.302920Z","submitted_at":"2024-06-12T04:09:44Z","title":"VALL-E R: Robust and Efficient Zero-Shot Text-to-Speech Synthesis via Monotonic Alignment","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.07855","snapshot_observed_at":"2026-08-06T22:19:01.038507Z","title":"V ALL-E R: Robust and Efficient Zero-Shot Text-to-Speech Synthesis via Monotonic Alignment,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:01.038507Z"},"links":{"cited_paper":"/paper/2406.07855","citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:8a99b206e6f054959d45d6e49d540606e11378d6eeff2b7c97ecd60aee103cd5","observation_id":"da940102-74c5-4465-8666-46c1934a7c50","resolution":{"observed_at":"2026-08-06T22:19:01.038507Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:19:07.395485Z","title":"V ALL-T: Decoder-only generative transducer for robust and decoding-controllable text-to-speech,","venue":null,"work_id":"284fc422-c45c-444e-b177-5ed9bc44a751","year":2025},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:01.123238Z"},"links":{"citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:2a4b24e1499bc1005398bb948a817414ed09baca324d9249bcaf93eb5ba92470","observation_id":"8b915efd-50a3-4683-9d04-669e86c5673f","resolution":{"observed_at":"2026-08-06T22:19:07.508375Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03204","last_updated":"2024-05-19T21:34:28Z","snapshot_observed_at":"2026-08-13T00:37:59.310571Z","submitted_at":"2024-04-04T05:15:07Z","title":"RALL-E: Robust Codec Language Modeling with Chain-of-Thought Prompting for Text-to-Speech Synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.03204","snapshot_observed_at":"2026-08-06T22:19:01.200105Z","title":"RALL-E: Robust Codec Language Modelling with Chain-of- Thought Prompting for Text-to-Speech Synthesis,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:01.200105Z"},"links":{"cited_paper":"/paper/2404.03204","citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:6726545af264170d7e941b12e2ce95402a068ceb27978b653e6b8b2d9bdf3401","observation_id":"0b967284-e98e-45fc-b7af-9a6177e2998a","resolution":{"observed_at":"2026-08-06T22:19:01.200105Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.14411","last_updated":"2024-10-18T12:24:05Z","snapshot_observed_at":"2026-08-12T22:20:08.175893Z","submitted_at":"2024-10-18T12:24:05Z","title":"SNAC: Multi-Scale Neural Audio Codec","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.14411","snapshot_observed_at":"2026-08-06T22:19:01.314755Z","title":"SNAC: Multi-Scale Neural Audio Codec,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:01.314755Z"},"links":{"cited_paper":"/paper/2410.14411","citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:e4ed1d4098940b177a36686f2ca83127b5c3699bc4d7d98d452ee333585fafe8","observation_id":"2c9ca762-3128-47f8-95e8-b181b5323ef8","resolution":{"observed_at":"2026-08-06T22:19:01.314755Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.11630","last_updated":"2024-09-18T01:31:19Z","snapshot_observed_at":"2026-08-12T22:42:49.545405Z","submitted_at":"2024-09-18T01:31:19Z","title":"Speaking from Coarse to Fine: Improving Neural Codec Language Model via Multi-Scale Speech Coding and Generation","version":1},"cited_work":{"arxiv_id":"2409.11630","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.11630","snapshot_observed_at":"2026-08-06T22:19:04.252786Z","title":"Speaking from Coarse to Fine: Improving Neural Codec Language Model via Multi-Scale Speech Coding and Generation","venue":"cs.SD","work_id":"26853cf2-2163-4bc2-bd92-fd732a35bc3c","year":2024},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:01.383749Z"},"links":{"cited_paper":"/paper/2409.11630","citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:f5c9816b555cbf1a1a0c2c926a3e5c1a4e96a768b5526ca419c8e1ae0605eae2","observation_id":"9ff7cc4e-5338-474b-a2b5-c427656fce26","resolution":{"observed_at":"2026-08-06T22:19:04.337592Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:19:07.152348Z","title":"UniAudio 1.5: Large Language Model-Driven Audio Codec is A Few-Shot Audio Task Learner,","venue":null,"work_id":"7df96fa8-dd78-40bc-be9e-c8eeea7fe902","year":2024},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:01.499041Z"},"links":{"citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:2e771e48a8cd769ccd9b44706a82237faf203182b648799fc45002cb23625ccf","observation_id":"453300fe-b9e7-4766-bb5e-e29390984124","resolution":{"observed_at":"2026-08-06T22:19:07.260505Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:19:06.880829Z","title":"SpeechTokenizer: Unified Speech Tokenizer for Speech Language Models,","venue":null,"work_id":"a6f9721c-6d52-43f2-871a-98de3b7812f4","year":2024},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:01.620066Z"},"links":{"citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:6ae015f5645a0494a1e802e2bfad46417e67f5657e5768df4da8bd4ac283a95d","observation_id":"623308b0-5410-4c93-921d-76604959d8cc","resolution":{"observed_at":"2026-08-06T22:19:07.030523Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.00037","last_updated":"2024-10-02T09:11:45Z","snapshot_observed_at":"2026-07-30T10:21:14.474746Z","submitted_at":"2024-09-17T17:55:39Z","title":"Moshi: a speech-text foundation model for real-time dialogue","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.00037","snapshot_observed_at":"2026-08-06T22:19:01.760430Z","title":"Moshi: A Speech-Text Foundation Model for Real-Time Dialogue,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:01.760430Z"},"links":{"cited_paper":"/paper/2410.00037","citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:fea0ba674bd02aa6c22e697ca9274a34d2790237410c4b434801a6656fa193fa","observation_id":"541ac633-3efc-459c-82c6-140d69c2f425","resolution":{"observed_at":"2026-08-06T22:19:01.760430Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:19:06.712858Z","title":"AudioLM: A Language Modeling Approach to Audio Generation,","venue":null,"work_id":"0d402815-270a-43e4-8ed2-ae7d0a783b27","year":2023},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:01.801928Z"},"links":{"citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:6206d3271ee79883b069ad8d25248b13487a038639bb6d2dbcf34e0e6512985b","observation_id":"f31aca70-c936-420a-b30f-4a1e6c6ec694","resolution":{"observed_at":"2026-08-06T22:19:06.814846Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:19:06.438213Z","title":"Speak, Read and Prompt: High-Fidelity Text-to- Speech with Minimal Supervision,","venue":null,"work_id":"05111200-78f8-4b89-a195-0a3ab5bcdf26","year":2023},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:01.863234Z"},"links":{"citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:149417ce88f881e5e393f967d4c852c916303bd62eb9e90a527db451f28e2186","observation_id":"87d7d26a-7d45-4b9e-a157-954651f6f045","resolution":{"observed_at":"2026-08-06T22:19:06.546180Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:19:06.206342Z","title":"MaskGCT: Zero-Shot Text-to-Speech with Masked Generative Codec Transformer,","venue":null,"work_id":"7f58e337-77d2-40c2-8619-e38bb041351a","year":2025},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:01.965874Z"},"links":{"citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:d28c7879e315e4558c2424e1dc24685c952d4e8d2f8a6a82426a38650098ae8d","observation_id":"212b35dc-657a-48f6-ae02-a3db385f2a66","resolution":{"observed_at":"2026-08-06T22:19:06.308846Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.10774","last_updated":"2024-06-14T23:32:32Z","snapshot_observed_at":"2026-07-06T17:17:56.276857Z","submitted_at":"2024-01-19T15:48:40Z","title":"Medusa: Simple LLM Inference Acceleration Framework with Multiple Decoding Heads","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.10774","snapshot_observed_at":"2026-08-06T22:19:02.032915Z","title":"Medusa: Simple llm inference acceleration framework with multiple decoding heads,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:02.032915Z"},"links":{"cited_paper":"/paper/2401.10774","citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:0d1615d4b876ce16c3b48cf51e7667e65ff78db2394b9297c241ed15fb6dece6","observation_id":"021b628d-e59d-41e5-9d24-abc2e64e2af3","resolution":{"observed_at":"2026-08-06T22:19:02.032915Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.19737","last_updated":"2024-04-30T17:33:57Z","snapshot_observed_at":"2026-08-12T13:06:04.804423Z","submitted_at":"2024-04-30T17:33:57Z","title":"Better & Faster Large Language Models via Multi-token Prediction","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.19737","snapshot_observed_at":"2026-08-06T22:19:02.090093Z","title":"Better & faster large language models via multi-token prediction,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:02.090093Z"},"links":{"cited_paper":"/paper/2404.19737","citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:435ae753a988ca4844689a519169a545c5c1a5a352dc1900f7406063d619c5f1","observation_id":"36d5de7f-a8ec-4f5a-bd80-308858dd3cb8","resolution":{"observed_at":"2026-08-06T22:19:02.090093Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-08-11T01:48:59.557045Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-06T22:19:02.210858Z","title":"Deepseek-v3 technical report,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:02.210858Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:05f445b731e9a37c29e9a53cc40363ccad5b6ef737f72c3e66173b06a700a884","observation_id":"01db0498-5f06-4f96-ac2c-35d8cbbe18d4","resolution":{"observed_at":"2026-08-06T22:19:02.210858Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:19:02.317947Z","title":"Training language models to follow instructions with human feedback,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:02.317947Z"},"links":{"citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:a0ff1e82e75284a7cf09a1f3e8c9856a85e850216d04197bae80a987bb45853a","observation_id":"034f4c16-c96a-4c01-8782-c94c93b10491","resolution":{"observed_at":"2026-08-06T22:19:02.317947Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-06T22:19:02.460098Z","title":"Proximal policy optimization algorithms,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:02.460098Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:e9bb78735545c35a0f77ff29ba85e591a4276909199b8d22090f94b8a2b0c528","observation_id":"b1f83a33-32f4-4d46-ad4d-407412ceb971","resolution":{"observed_at":"2026-08-06T22:19:02.460098Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:19:02.578605Z","title":"Direct preference optimization: Your language model is secretly a reward model,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:02.578605Z"},"links":{"citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:e86dcacc626f0717dd0c430ad01ae5756b927846ad2c8467ec27e9c2dee07690","observation_id":"8b2ec461-051b-48dd-a584-65794bf75d01","resolution":{"observed_at":"2026-08-06T22:19:02.578605Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-08-13T15:58:13.809876Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-06T22:19:02.652495Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:02.652495Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:12d077685dccdc9d0102a3303e663e301f079e49a627abeaa1e4bc04e0c91fb7","observation_id":"ed5e40cf-f30d-403d-b151-12f36ac6cea7","resolution":{"observed_at":"2026-08-06T22:19:02.652495Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.01834","last_updated":"2025-05-27T16:17:52Z","snapshot_observed_at":"2026-08-13T23:53:43.973682Z","submitted_at":"2024-11-04T06:07:53Z","title":"Align-SLM: Textless Spoken Language Models with Reinforcement Learning from AI Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.01834","snapshot_observed_at":"2026-08-06T22:19:02.738861Z","title":"Align-SLM: Textless Spoken Language Models with Reinforcement Learning from AI Feedback,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:02.738861Z"},"links":{"cited_paper":"/paper/2411.01834","citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:aafd09d7537ee86b7f7f54c3e1f7764eb2cba92134b505a15d7b4dabc10becf9","observation_id":"b38bdcb0-3913-450b-8c11-8b7e27459dd7","resolution":{"observed_at":"2026-08-06T22:19:02.738861Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.02243","last_updated":"2024-07-02T13:04:04Z","snapshot_observed_at":"2026-08-12T23:30:26.059993Z","submitted_at":"2024-07-02T13:04:04Z","title":"Robust Zero-Shot Text-to-Speech Synthesis with Reverse Inference Optimization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.02243","snapshot_observed_at":"2026-08-06T22:19:02.830612Z","title":"Robust zero-shot text-to-speech synthesis with reverse inference optimization,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:02.830612Z"},"links":{"cited_paper":"/paper/2407.02243","citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:cae2b42210378e0d1c8fae1188204822bac1b7feea79e03da34dd1d6ab38f184","observation_id":"76cb5b0b-a16d-4a52-87f3-b884024da7cf","resolution":{"observed_at":"2026-08-06T22:19:02.830612Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.00654","last_updated":"2024-06-02T07:54:33Z","snapshot_observed_at":"2026-08-13T16:07:28.259490Z","submitted_at":"2024-06-02T07:54:33Z","title":"Enhancing Zero-shot Text-to-Speech Synthesis with Human Feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.00654","snapshot_observed_at":"2026-08-06T22:19:02.888700Z","title":"Enhancing zero-shot text-to- speech synthesis with human feedback,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:02.888700Z"},"links":{"cited_paper":"/paper/2406.00654","citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:13dcbbb4a48c4ab67e11e987c8e21780922a12c6c969a1b674248ecb28aa7898","observation_id":"aff5959b-423b-4aed-a646-6c0e663d169a","resolution":{"observed_at":"2026-08-06T22:19:02.888700Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:19:05.901287Z","title":"SpeechAlign: Aligning speech generation to human preferences,","venue":null,"work_id":"69c7f58b-5891-4bf8-b89c-a846a18502d0","year":2024},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:02.963953Z"},"links":{"citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:3ac5d1772281acc41ead2265a1ad3285f1c3758b757fb62dbea4bd9308a3008a","observation_id":"576192fd-545d-423f-a2fd-742d69e6552d","resolution":{"observed_at":"2026-08-06T22:19:06.010259Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:19:03.066762Z","title":"Fine-grained preference optimization improves zero-shot text-to-speech,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:03.066762Z"},"links":{"citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:ee6d16fa23c1d07a0f9d088424787fbf61c64ceb4e6489944cc044ef06867b43","observation_id":"71a4a6ff-5a9b-4326-9b1f-18cdfee7b83c","resolution":{"observed_at":"2026-08-06T22:19:03.066762Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-12T23:47:26.090796Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-08-06T22:19:03.126226Z","title":"V ALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:03.126226Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:21cc322dfa2f64b980e2838a43b613cbb4e968b4029822c20440fe99ae83df17","observation_id":"9a995078-6c79-4198-9ada-73e1eedc2b32","resolution":{"observed_at":"2026-08-06T22:19:03.126226Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:19:05.698777Z","title":"UniCATS: A Unified Context-Aware Text-to-Speech Framework with Contextual VQ-Diffusion and V ocoding,","venue":null,"work_id":"fc680e15-9224-42d0-bbab-efdb83b48f98","year":2024},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:03.250253Z"},"links":{"citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:145d837ebdda74601c724630f75c9514453d20b224bcca9e6690a17ace5fbad0","observation_id":"f9935a8d-bf66-4312-a1fc-ca1e4ab6f111","resolution":{"observed_at":"2026-08-06T22:19:05.793544Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.15764","last_updated":"2025-05-21T16:46:32Z","snapshot_observed_at":"2026-08-12T22:48:20.055495Z","submitted_at":"2024-10-21T08:23:31Z","title":"LSCodec: Low-Bitrate and Speaker-Decoupled Discrete Speech Codec","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.15764","snapshot_observed_at":"2026-08-06T22:19:03.356798Z","title":"LSCodec: Low-Bitrate and Speaker-Decoupled Discrete Speech Codec,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:03.356798Z"},"links":{"cited_paper":"/paper/2410.15764","citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:97cb4cf01b5c89180f040b828867a497aabbe64d6a3185585c90d8f79584b565","observation_id":"e51de082-c27f-4ccf-9cef-17c1e6e69518","resolution":{"observed_at":"2026-08-06T22:19:03.356798Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:19:05.549527Z","title":"WavLM: Large-Scale Self-Supervised Pre-Training for Full Stack Speech Processing,","venue":null,"work_id":"124dc65d-737d-46fe-8edc-4f99f0777dbc","year":2022},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:03.471288Z"},"links":{"citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:14fbc60a8909da837fb17a90fa1ceb29c2986317d3b7a82ad708f69164ba0cdd","observation_id":"65daf919-b05e-4438-8033-4f3a98cb4407","resolution":{"observed_at":"2026-08-06T22:19:05.611424Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:19:03.560664Z","title":"Sigmoid-weighted linear units for neural network function approximation in reinforcement learning,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:03.560664Z"},"links":{"citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:41543dd7a74229f6f8251b3a03c18279292f24dc8e01ae5b51d4e7d1f3538fa9","observation_id":"d8c83b8d-7981-4eae-a62d-7fefc3f3a776","resolution":{"observed_at":"2026-08-06T22:19:03.560664Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:19:05.234097Z","title":"UTMOS: UTokyo- SaruLab System for V oiceMOS Challenge 2022,","venue":null,"work_id":"ed7b522f-81ed-4493-ba7d-44c66d3e0df0","year":2022},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:03.693568Z"},"links":{"citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:5affafd9b58f246eeeaec9695903945916c48b0e052e9b3d7e21d9d43a9089b1","observation_id":"d2d0d98b-d896-40cd-99c4-b34e37f99f92","resolution":{"observed_at":"2026-08-06T22:19:05.438167Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:19:04.973446Z","title":"Robust Speech Recognition via Large-Scale Weak Supervision,","venue":null,"work_id":"7dbd8584-458a-485f-80c0-5007155adecb","year":2023},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:03.772752Z"},"links":{"citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:84799e39d0f242eb4e2aa63154d2d6f4672945aae1e2993f94c5610565b929bd","observation_id":"75ad788d-482b-47b3-9388-04dc6c7cf032","resolution":{"observed_at":"2026-08-06T22:19:05.082403Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.11230","last_updated":"2024-04-10T02:35:38Z","snapshot_observed_at":"2026-08-13T05:47:38.366792Z","submitted_at":"2023-10-17T13:01:10Z","title":"Zipformer: A faster and better encoder for automatic speech recognition","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.11230","snapshot_observed_at":"2026-08-06T22:19:03.875806Z","title":"Zipformer: A faster and better encoder for automatic speech recognition,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:03.875806Z"},"links":{"cited_paper":"/paper/2310.11230","citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:8a49a95fc926ba704b2feeec43db1db73ae2c253783f8cc73e8d457a5175a596","observation_id":"5835c07c-950d-4118-957f-594a85fa128b","resolution":{"observed_at":"2026-08-06T22:19:03.875806Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy"},"reference_resolution":{"displayed":50,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":30,"verified_exact":2,"verified_fuzzy":18},"total_outbound_references":50},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 50 of 50 outbound references and 2 inbound Pith citation observations for arXiv:2506.22023."}