{"as_of":"2026-08-13T11:04:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:81969e3870ac62271314d468762a5c6021624876bc271c57ea50470920d50575","coverage":[{"denominator":24,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":24,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T19:32:24.151140Z","state":"measured"},{"denominator":31,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":31,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":7,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":7,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T13:21:57.611094Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-06-30T12:24:39.704893Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.06602","snapshot_observed_at":"2026-08-07T13:21:57.611094Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.22053","last_updated":"2025-08-05T07:47:10Z","snapshot_observed_at":"2026-08-10T06:47:29.527434Z","submitted_at":"2025-05-28T07:23:53Z","title":"AudioGenie: A Training-Free Multi-Agent Framework for Diverse Multimodality-to-Multiaudio Generation","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T13:21:57.611094Z"},"links":{"cited_paper":"/paper/2412.06602","citing_paper":"/paper/2505.22053"},"observation_digest":"sha256:d64e3c3d0d0fb05f0201ed5c2b8260d2827bec4c5f8d65563f1632fd19f79eda","observation_id":"e230f18e-81fa-4d4d-8444-6c259f26b889","resolution":{"observed_at":"2026-08-07T13:21:57.611094Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.06602","snapshot_observed_at":"2026-08-06T21:25:27.548842Z","title":"Towards controllable speech synthesis in the era of large language models: A survey,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00227","last_updated":"2025-06-30T19:52:32Z","snapshot_observed_at":"2026-08-08T22:36:28.237803Z","submitted_at":"2025-06-30T19:52:32Z","title":"Investigating Stochastic Methods for Prosody Modeling in Speech Synthesis","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T21:25:27.548842Z"},"links":{"cited_paper":"/paper/2412.06602","citing_paper":"/paper/2507.00227"},"observation_digest":"sha256:d52ade97a67c583b35aab35f8b50a5fe5b9f7c6f31b5443729e864da516f2561","observation_id":"49542191-d96c-4e3d-b266-6d82137c50fe","resolution":{"observed_at":"2026-08-06T21:25:27.548842Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.06602","snapshot_observed_at":"2026-08-05T11:25:29.118241Z","title":"Towards controllable speech synthesis in the era of large language models: A survey,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.02859","last_updated":"2025-09-02T22:11:29Z","snapshot_observed_at":"2026-08-05T11:25:28.709116Z","submitted_at":"2025-09-02T22:11:29Z","title":"Speech DF Arena: A Leaderboard for Speech DeepFake Detection Models","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-05T11:25:29.118241Z"},"links":{"cited_paper":"/paper/2412.06602","citing_paper":"/paper/2509.02859"},"observation_digest":"sha256:4d5d1e2e81fc6a2d6db3df7b3bb5268163a3f3752504e28319529c51d7de08d1","observation_id":"da47824a-1952-498a-937c-9ff0fa131f91","resolution":{"observed_at":"2026-08-05T11:25:29.118241Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"cited_work":{"arxiv_id":"2412.06602","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.06602","snapshot_observed_at":"2026-06-30T12:24:39.704893Z","title":"Towards con- trollable speech synthesis in the era of large language models: A survey","venue":null,"work_id":"3abceb1a-343d-4a40-a369-ddaad0b3872f","year":2021},"citing_paper":{"arxiv_id":"2510.06201","last_updated":"2026-05-04T15:09:53Z","snapshot_observed_at":"2026-08-06T05:09:49.220327Z","submitted_at":"2025-10-07T17:54:12Z","title":"TokenChain: A Discrete Speech Chain via Semantic Token Modeling","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-18T09:14:58.542628Z"},"links":{"cited_paper":"/paper/2412.06602","citing_paper":"/paper/2510.06201"},"observation_digest":"sha256:8b94138cfa0aed624fb524764137e6da867cfce34625dfc07f83be4d588da038","observation_id":"80d5efeb-243f-41f0-8ad0-0e4b99600dbe","resolution":{"observed_at":"2026-05-18T09:16:09.518750Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.06602","snapshot_observed_at":"2026-08-04T11:07:41.013784Z","title":"Guanrou Yang, Chen Yang, Qian Chen, and 1 others","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2510.06927","last_updated":"2026-05-25T16:15:03Z","snapshot_observed_at":"2026-08-12T04:22:43.216951Z","submitted_at":"2025-10-08T12:07:57Z","title":"Position: Towards Responsible Evaluation for Text-to-Speech","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-04T11:07:41.013784Z"},"links":{"cited_paper":"/paper/2412.06602","citing_paper":"/paper/2510.06927"},"observation_digest":"sha256:8ae4c16dd546c45ad5276be4d8d41979547be43088c6628ec03aedf322baed61","observation_id":"6e27a746-8fdd-4de7-b05f-cdd5558197e6","resolution":{"observed_at":"2026-08-04T11:07:41.013784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"cited_work":{"arxiv_id":"2412.06602","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.06602","snapshot_observed_at":"2026-06-30T12:24:39.704893Z","title":"Towards con- trollable speech synthesis in the era of large language models: A survey","venue":null,"work_id":"3abceb1a-343d-4a40-a369-ddaad0b3872f","year":2021},"citing_paper":{"arxiv_id":"2604.08363","last_updated":"2026-04-09T15:27:22Z","snapshot_observed_at":"2026-08-13T09:39:50.855408Z","submitted_at":"2026-04-09T15:27:22Z","title":"CapTalk: Unified Voice Design for Single-Utterance and Dialogue Speech Generation","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-10T17:15:28.918204Z"},"links":{"cited_paper":"/paper/2412.06602","citing_paper":"/paper/2604.08363"},"observation_digest":"sha256:7a1f2c624f45c5e7f84ff3fdda3be91d54596bfa034985b2b0596041bf6eb424","observation_id":"f2c35a8c-70a6-4ef1-8039-8338f2332ce1","resolution":{"observed_at":"2026-05-11T07:16:00.009111Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"cited_work":{"arxiv_id":"2412.06602","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.06602","snapshot_observed_at":"2026-06-30T12:24:39.704893Z","title":"Towards con- trollable speech synthesis in the era of large language models: A survey","venue":null,"work_id":"3abceb1a-343d-4a40-a369-ddaad0b3872f","year":2021},"citing_paper":{"arxiv_id":"2605.24618","last_updated":"2026-05-23T15:01:28Z","snapshot_observed_at":"2026-08-03T19:22:55.968438Z","submitted_at":"2026-05-23T15:01:28Z","title":"FC-TTS: Style and Timbre Control in Zero-Shot Text-to-Speech with Disentangled Speech Representations","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-30T12:20:43.383513Z"},"links":{"cited_paper":"/paper/2412.06602","citing_paper":"/paper/2605.24618"},"observation_digest":"sha256:a9830ef412849320c9deb60234520984077c2eb6d0dcb90c0ee8c07b302fd063","observation_id":"231ab4b1-d86d-4661-9316-3f6467049e24","resolution":{"observed_at":"2026-06-30T12:24:39.706455Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2412.06602/citation-record","integrity":"/paper/2412.06602/integrity","json":"/paper/2412.06602/citation-record.json","paper":"/paper/2412.06602"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T19:32:24.446984Z","title":"• 2 points: It loosely follows the instruc- tions but misses key elements or timing in parts","venue":null,"work_id":"c29751cd-6bed-45bf-8e20-0cda3fc698b3","year":null},"citing_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-11T19:32:24.141778Z"},"links":{"citing_paper":"/paper/2412.06602"},"observation_digest":"sha256:b4fd8eb6e7b82d83154587c03eff6fa3dba65f98bd2dc9a2dda4145cf6aa3e0c","observation_id":"d6878a65-541c-486d-a016-74a4b6ada6ab","resolution":{"observed_at":"2026-08-11T19:32:24.452307Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T19:32:24.427478Z","title":"• 2 points: Noticeably synthetic; some un- natural artifacts remain","venue":null,"work_id":"11760a28-a4e9-428f-ac38-1ee9290be0d1","year":null},"citing_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-11T19:32:24.146371Z"},"links":{"citing_paper":"/paper/2412.06602"},"observation_digest":"sha256:20604ccc7c197dac2968337a26aca43a6ddd6b614d40f8e6eb11b120e63116b6","observation_id":"3a9db7a4-81d2-44aa-bbb3-6bad7270ceb7","resolution":{"observed_at":"2026-08-11T19:32:24.433638Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T19:32:24.409507Z","title":"{transcript}","venue":null,"work_id":"a59db512-f17f-453a-927c-ee3476869745","year":null},"citing_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-11T19:32:24.151140Z"},"links":{"citing_paper":"/paper/2412.06602"},"observation_digest":"sha256:a4917b28647a9f4d2d33cf42004e59ebceed6ebfbff33fb840259db927352d1e","observation_id":"6571bb23-6e58-4ae1-851e-97d0c7b089ba","resolution":{"observed_at":"2026-08-11T19:32:24.415747Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T19:32:24.558991Z","title":"In Pro- ceedings of the 32nd ACM International Conference on Multimedia, pages 1255–1264","venue":null,"work_id":"b17f44be-799e-4718-be2c-1a519c7fb9d7","year":2024},"citing_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-11T19:32:24.061207Z"},"links":{"citing_paper":"/paper/2412.06602"},"observation_digest":"sha256:1cf999736e055eebc4a537d8858203210f2e81f647ee9bd3b0302004ef0df5a4","observation_id":"bc464c2f-4265-4bd8-8681-969f8a033022","resolution":{"observed_at":"2026-08-11T19:32:24.564296Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.12454","last_updated":"2023-11-27T12:26:32Z","snapshot_observed_at":"2026-08-13T05:20:35.634746Z","submitted_at":"2023-11-21T09:07:11Z","title":"HierSpeech++: Bridging the Gap between Semantic and Acoustic Representation of Speech by Hierarchical Variational Inference for Zero-shot Speech Synthesis","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.12454","snapshot_observed_at":"2026-08-11T19:32:24.075261Z","title":"Advances in Neural Information Processing Systems, 36","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-11T19:32:24.075261Z"},"links":{"cited_paper":"/paper/2311.12454","citing_paper":"/paper/2412.06602"},"observation_digest":"sha256:12561ac73c583de6701feb9b4561f9e8fe6caab5632921ca01b3bab1df4fab31","observation_id":"ecc71c65-57fa-416a-ba47-2d1417464cdc","resolution":{"observed_at":"2026-08-11T19:32:24.075261Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T19:32:24.543587Z","title":null,"venue":null,"work_id":"d0466755-5538-404d-bc6f-8051641741f7","year":null},"citing_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-11T19:32:24.080857Z"},"links":{"citing_paper":"/paper/2412.06602"},"observation_digest":"sha256:37af856526069c2bbf0e01002197b26723cdf9021f062e6982a93ffdf6091de2","observation_id":"59212e56-b630-4a0c-a627-112f10d8710f","resolution":{"observed_at":"2026-08-11T19:32:24.548472Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-08-13T04:28:02.323611Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-11T19:32:24.085937Z","title":"In Proceedings of the 31st ACM In- ternational Conference on Multimedia, pages 2829– 2837","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-11T19:32:24.085937Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2412.06602"},"observation_digest":"sha256:535b1a0ed1b7cecb94875b0b26db8b7609c5244bc21cd945a2b5ccca7f406bd6","observation_id":"e8d21e48-135a-49c3-974e-ee72316c30c3","resolution":{"observed_at":"2026-08-11T19:32:24.085937Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T19:32:24.529037Z","title":"Advances in Neural Information Processing Systems, 36:53728– 53741","venue":null,"work_id":"32ebd7fc-13b4-46d1-8b7f-c625a25f5c79","year":2020},"citing_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-11T19:32:24.091663Z"},"links":{"citing_paper":"/paper/2412.06602"},"observation_digest":"sha256:a27373d6557d7bb02bdd3811e066226db9beb8f675da77a1957964827ca742f3","observation_id":"1a48b56a-ea5f-44e5-b3c5-c91699a4fef8","resolution":{"observed_at":"2026-08-11T19:32:24.533789Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.15561","last_updated":"2021-07-23T12:32:52Z","snapshot_observed_at":"2026-08-08T12:28:23.489422Z","submitted_at":"2021-06-29T16:50:51Z","title":"A Survey on Neural Speech Synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.15561","snapshot_observed_at":"2026-08-11T19:32:24.106056Z","title":"In IEEE Spoken Lan- guage Technology Workshop, pages 595–602","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-11T19:32:24.106056Z"},"links":{"cited_paper":"/paper/2106.15561","citing_paper":"/paper/2412.06602"},"observation_digest":"sha256:2acdfcfb1b030a9a3b690e7497b462002a768ee612fdf3285f6314486f29c725","observation_id":"bd28493a-eff6-428c-b292-10e6f4bb2c26","resolution":{"observed_at":"2026-08-11T19:32:24.106056Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03204","last_updated":"2024-05-19T21:34:28Z","snapshot_observed_at":"2026-08-13T00:37:59.310571Z","submitted_at":"2024-04-04T05:15:07Z","title":"RALL-E: Robust Codec Language Modeling with Chain-of-Thought Prompting for Text-to-Speech Synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.03204","snapshot_observed_at":"2026-08-11T19:32:24.110824Z","title":"arXiv preprint arXiv:2404.03204","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-11T19:32:24.110824Z"},"links":{"cited_paper":"/paper/2404.03204","citing_paper":"/paper/2412.06602"},"observation_digest":"sha256:8a14c7684c933a99b578d55c5a7a9e6cabe57ab728dbf3da1337e7ddb6a27983","observation_id":"4b4e9803-615b-4b18-994a-37972b25e66e","resolution":{"observed_at":"2026-08-11T19:32:24.110824Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.03926","last_updated":"2023-03-07T14:31:55Z","snapshot_observed_at":"2026-08-06T04:55:08.186019Z","submitted_at":"2023-03-07T14:31:55Z","title":"Speak Foreign Languages with Your Own Voice: Cross-Lingual Neural Codec Language Modeling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.03926","snapshot_observed_at":"2026-08-11T19:32:24.120019Z","title":"In IEEE International Conference on Acoustics, Speech and Signal Processing, pages 6945–6949","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-11T19:32:24.120019Z"},"links":{"cited_paper":"/paper/2303.03926","citing_paper":"/paper/2412.06602"},"observation_digest":"sha256:14d36706ba3e63289d4136c1528728fcde5f6ffc2c1213e0adaf4803f6d83047","observation_id":"0fecd5ed-31e5-42fb-a104-c5c451a205fd","resolution":{"observed_at":"2026-08-11T19:32:24.120019Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T19:32:24.498107Z","title":null,"venue":null,"work_id":"73673bad-c716-4bb5-8b72-ddb5cb85d373","year":null},"citing_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-11T19:32:24.128863Z"},"links":{"citing_paper":"/paper/2412.06602"},"observation_digest":"sha256:d6d6fdeedac14309696a64bd06c7c89694d3f38152d12f6b5dd5b92d892dea7c","observation_id":"bfd49210-b857-46ad-a415-62ea5a818b66","resolution":{"observed_at":"2026-08-11T19:32:24.502947Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T19:32:24.480374Z","title":"Speech vocoder is the last com- ponent that converts the intermediate acoustic fea- tures into a waveform that can be played back","venue":null,"work_id":"322d93e1-4940-4d79-a276-0f03cd6a896e","year":2021},"citing_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-11T19:32:24.133064Z"},"links":{"citing_paper":"/paper/2412.06602"},"observation_digest":"sha256:b28a8ea593ddf8a0d27cf5c941c612903c2950c7dd691fe2ea9a153bdfd0cdef","observation_id":"ca8827d7-9fe1-4990-aa62-bd9ad8a7474a","resolution":{"observed_at":"2026-08-11T19:32:24.485242Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.11002","last_updated":"2025-08-12T15:55:09Z","snapshot_observed_at":"2026-08-08T05:59:50.599815Z","submitted_at":"2025-04-15T09:19:44Z","title":"Dopamine Audiobook: A Training-free MLLM Agent for Emotional and Immersive Audiobook Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.11002","snapshot_observed_at":"2026-08-11T19:32:24.096350Z","title":"In 2001 IEEE International Conference on Acoustics, Speech, and Signal Pro- cessing","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"reference_index":2001,"source":"pdf_text","source_observed_at":"2026-08-11T19:32:24.096350Z"},"links":{"cited_paper":"/paper/2504.11002","citing_paper":"/paper/2412.06602"},"observation_digest":"sha256:fc5f2fce7002c944fcf23605e7204b07dd237216b13e61febf2c7d4a605cdc5f","observation_id":"909a04ca-16c9-4b60-94ea-d2a8d2d4ec1d","resolution":{"observed_at":"2026-08-11T19:32:24.096350Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T19:32:24.463993Z","title":"A lower MCD value in- dicates a higher similarity between synthesized and reference speech, meaning better speech synthesis quality","venue":null,"work_id":"4362c73f-49d5-4360-a4df-e6b897959214","year":2020},"citing_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"reference_index":2008,"source":"pdf_text","source_observed_at":"2026-08-11T19:32:24.136896Z"},"links":{"citing_paper":"/paper/2412.06602"},"observation_digest":"sha256:21f841c0305ddf78b2fa4d593829c55143511dea264b50fb7cd0331f3f394c43","observation_id":"d82ed43d-ac82-46bb-84af-9682cc594194","resolution":{"observed_at":"2026-08-11T19:32:24.469901Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.10157","last_updated":"2024-09-16T10:41:36Z","snapshot_observed_at":"2026-08-12T22:44:12.350277Z","submitted_at":"2024-09-16T10:41:36Z","title":"Emo-DPO: Controllable Emotional Speech Synthesis through Direct Preference Optimization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.10157","snapshot_observed_at":"2026-08-11T19:32:24.042006Z","title":"In IEEE Interna- tional Conference on Acoustics, Speech and Signal Processing, pages 4475–4479","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"reference_index":2015,"source":"pdf_text","source_observed_at":"2026-08-11T19:32:24.042006Z"},"links":{"cited_paper":"/paper/2409.10157","citing_paper":"/paper/2412.06602"},"observation_digest":"sha256:5dd89f9237bf2e38790358d70ce46738f637547b2ad178715c1d0fe85e9ee15e","observation_id":"10ef6b29-1f0f-4fea-9e65-4069343e7774","resolution":{"observed_at":"2026-08-11T19:32:24.042006Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T19:32:24.514176Z","title":"speak calmly","venue":null,"work_id":"a9a2b747-7e93-4ff8-9158-b9c895410c93","year":2017},"citing_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"reference_index":2016,"source":"pdf_text","source_observed_at":"2026-08-11T19:32:24.124409Z"},"links":{"citing_paper":"/paper/2412.06602"},"observation_digest":"sha256:9af04b1fe3035970fcd7ea54ed286d017bbda16ed1a4bd39f7553ed6c27c8196","observation_id":"a61ed7bf-e415-42a3-a621-40b685f5b6ef","resolution":{"observed_at":"2026-08-11T19:32:24.518923Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05407","last_updated":"2024-07-09T07:42:51Z","snapshot_observed_at":"2026-08-10T17:49:50.848957Z","submitted_at":"2024-07-07T15:16:19Z","title":"CosyVoice: A Scalable Multilingual Zero-shot Text-to-speech Synthesizer based on Supervised Semantic Tokens","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.05407","snapshot_observed_at":"2026-08-11T19:32:24.035818Z","title":"In International Conference on Learning Representations","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"reference_index":2018,"source":"pdf_text","source_observed_at":"2026-08-11T19:32:24.035818Z"},"links":{"cited_paper":"/paper/2407.05407","citing_paper":"/paper/2412.06602"},"observation_digest":"sha256:48c4ef1d24149ae683b6eac5027ad1233d4afc69abae139d71a58bf328cc0ba1","observation_id":"33f32af4-c5f0-41bf-aa7b-08609e27b16f","resolution":{"observed_at":"2026-08-11T19:32:24.035818Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.11946","last_updated":"2025-02-18T07:29:10Z","snapshot_observed_at":"2026-08-12T20:35:16.024701Z","submitted_at":"2025-02-17T15:58:56Z","title":"Step-Audio: Unified Understanding and Generation in Intelligent Speech Interaction","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.11946","snapshot_observed_at":"2026-08-11T19:32:24.048348Z","title":"In ICASSP 2019 - 2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pages 5901–5905","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-11T19:32:24.048348Z"},"links":{"cited_paper":"/paper/2502.11946","citing_paper":"/paper/2412.06602"},"observation_digest":"sha256:897a923f240d342813745880d5e89a7f1b249d14e7ecd334c6025d5794848384","observation_id":"02937b2f-4361-4dd5-b3b2-0ff753baa0d2","resolution":{"observed_at":"2026-08-11T19:32:24.048348Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.13893","last_updated":"2024-08-28T07:16:37Z","snapshot_observed_at":"2026-08-12T22:57:49.226206Z","submitted_at":"2024-08-25T17:07:39Z","title":"SimpleSpeech 2: Towards Simple and Efficient Text-to-Speech with Flow-based Scalar Latent Transformer Diffusion Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.13893","snapshot_observed_at":"2026-08-11T19:32:24.115348Z","title":"In IEEE Interna- tional Conference on Acoustics, Speech and Signal Processing, pages 6199–6203","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-11T19:32:24.115348Z"},"links":{"cited_paper":"/paper/2408.13893","citing_paper":"/paper/2412.06602"},"observation_digest":"sha256:a3192b0b3003e25951d4395260c200ff3105c4831434e85aa32a2ab9fa44ed4b","observation_id":"8d23af91-f43b-42e5-8a7c-f30840e33a93","resolution":{"observed_at":"2026-08-11T19:32:24.115348Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07333","last_updated":"2024-01-14T17:43:55Z","snapshot_observed_at":"2026-08-13T04:43:10.925367Z","submitted_at":"2024-01-14T17:43:55Z","title":"ELLA-V: Stable Neural Codec Language Modeling with Alignment-guided Sequence Reordering","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.07333","snapshot_observed_at":"2026-08-11T19:32:24.101443Z","title":"In Conference of the International Speech Communication Association, pages 2756–2760","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-11T19:32:24.101443Z"},"links":{"cited_paper":"/paper/2401.07333","citing_paper":"/paper/2412.06602"},"observation_digest":"sha256:8f3053fbe1952cd608069bbeff9340b4bc0414b156c1e8731f9777a84d6ca78a","observation_id":"b0ae0045-a581-40fc-a133-ecb6925a0a34","resolution":{"observed_at":"2026-08-11T19:32:24.101443Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10550","last_updated":"2023-07-20T03:28:06Z","snapshot_observed_at":"2026-08-13T10:52:11.579798Z","submitted_at":"2023-07-20T03:28:06Z","title":"SC VALL-E: Style-Controllable Zero-Shot Text to Speech Synthesizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.10550","snapshot_observed_at":"2026-08-11T19:32:24.067231Z","title":"In IEEE International Conference on Acoustics, Speech and Signal Processing, pages 1–5","venue":null,"work_id":null,"year":1999},"citing_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-11T19:32:24.067231Z"},"links":{"cited_paper":"/paper/2307.10550","citing_paper":"/paper/2412.06602"},"observation_digest":"sha256:e2bb26f55025a41353c7f4162549d24f4dec5abfc4a0ea7dddd8852ac99fe438","observation_id":"b9a09387-1d28-43f3-a61f-51d1074a3f8d","resolution":{"observed_at":"2026-08-11T19:32:24.067231Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12498","last_updated":"2025-06-22T16:51:47Z","snapshot_observed_at":"2026-08-11T13:59:15.293616Z","submitted_at":"2024-12-17T03:02:05Z","title":"Hierarchical Control of Emotion Rendering in Speech Synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.12498","snapshot_observed_at":"2026-08-11T19:32:24.055265Z","title":"arXiv preprint arXiv:2412.12498","venue":null,"work_id":null,"year":1975},"citing_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-11T19:32:24.055265Z"},"links":{"cited_paper":"/paper/2412.12498","citing_paper":"/paper/2412.06602"},"observation_digest":"sha256:4bb32d0c930c7d19534f6e3e96bf27cb6e086355acfb6a38468d1afb5eba5559","observation_id":"99ae3806-5571-4a12-a529-2c9c05b018ec","resolution":{"observed_at":"2026-08-11T19:32:24.055265Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15649","last_updated":"2024-12-20T08:05:55Z","snapshot_observed_at":"2026-08-12T05:23:04.133755Z","submitted_at":"2024-12-20T08:05:55Z","title":"SLAM-Omni: Timbre-Controllable Voice Interaction System with Single-Stage Training","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15649","snapshot_observed_at":"2026-08-11T19:32:24.029868Z","title":"In IEEE International Conference on Acoustics, Speech and Signal Processing, pages 1–5","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","version":3},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-11T19:32:24.029868Z"},"links":{"cited_paper":"/paper/2412.15649","citing_paper":"/paper/2412.06602"},"observation_digest":"sha256:a554d130dc9a17cc4a7bb61935a0926b2e00c27be9573cb482232c38a3419746","observation_id":"8a279d26-5421-4784-a928-f7085b96354b","resolution":{"observed_at":"2026-08-11T19:32:24.029868Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2412.06602","last_updated":"2025-08-25T07:30:54Z","latest_version":3,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-11T19:26:50.636660Z","submitted_at":"2024-12-09T15:50:25Z","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey"},"reference_resolution":{"displayed":24,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":16,"verified_exact":0,"verified_fuzzy":7},"total_outbound_references":24},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 24 of 24 outbound references and 7 inbound Pith citation observations for arXiv:2412.06602."}