{"as_of":"2026-08-13T01:05:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:19aad80c63b3095922840c8bba667f9212fec078f70bc8cd56d90c44c46536ed","coverage":[{"denominator":52,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":52,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T18:51:21.143066Z","state":"measured"},{"denominator":53,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":53,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-12T06:34:41.77262+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T15:01:20.902986Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-11T11:21:00.973354Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"cited_work":{"arxiv_id":"2509.09631","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2509.09631","snapshot_observed_at":"2026-06-19T17:11:23.335684Z","title":"InAmericasNLP 2023, pages 206–219","venue":null,"work_id":"f3dae48a-8b12-4b55-a0de-a2840011e625","year":2023},"citing_paper":{"arxiv_id":"2604.13288","last_updated":"2026-04-14T20:32:29Z","snapshot_observed_at":"2026-08-11T14:34:41.499169Z","submitted_at":"2026-04-14T20:32:29Z","title":"Giving Voice to the Constitution: Low-Resource Text-to-Speech for Quechua and Spanish Using a Bilingual Legal Corpus","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T15:01:20.902986Z"},"links":{"cited_paper":"/paper/2509.09631","citing_paper":"/paper/2604.13288"},"observation_digest":"sha256:301305a71c778d90be339a92226414198bbab1cf87c15c37266cfd5cd3843035","observation_id":"9de43ea6-adf9-44e7-b6ee-7786e256cf53","resolution":{"observed_at":"2026-06-19T17:11:23.335684Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2509.09631/citation-record","integrity":"/paper/2509.09631/integrity","json":"/paper/2509.09631/citation-record.json","paper":"/paper/2509.09631"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:20.978381Z","title":", \" * write output.state after.block = add.period write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:20.978381Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:377625193ababd320c8c2feb1b9124698cc299a6376455bf06689c63186f6af5","observation_id":"3848f684-414d-466a-bab1-e23a2de7ab1e","resolution":{"observed_at":"2026-08-04T18:51:20.978381Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:20.982353Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:20.982353Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:fb08d558939d20224056dd6c228b29a1a6fde2e4212370004924d63ce4625c81","observation_id":"1efd47cc-1f74-4521-8257-c45412f15fb1","resolution":{"observed_at":"2026-08-04T18:51:20.982353Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:20.986264Z","title":"D.; Ho, J.; Tarlow, D.; and van den Berg, R","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:20.986264Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:1671d020cb4e1aa6f00aa63f5c0e1caef4cb1febd903decc6f33d1a38eae9eca","observation_id":"87b663b2-c4bc-449a-85d1-e13f5f08a19d","resolution":{"observed_at":"2026-08-04T18:51:20.986264Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:20.990362Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:20.990362Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:2b880a5c902ac667bd4eabed479372838946f6840409738f91ce26af2dc1fcf3","observation_id":"556249ed-9558-462e-bfb8-e63b992bfbec","resolution":{"observed_at":"2026-08-04T18:51:20.990362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:20.993955Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:20.993955Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:9a622550c2266d910473a0bd50894cd300d1840e1bc2c518e85fbef80c08fdca","observation_id":"cbcd4f63-1b2f-433a-b871-a58c5241350a","resolution":{"observed_at":"2026-08-04T18:51:20.993955Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-12T23:47:26.090796Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-08-04T18:51:20.997584Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:20.997584Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:ac5a7de19134eb0840e0094be4834350bc033d05efa935fb09237366a8b034d2","observation_id":"4dbb91b5-f808-4ded-a2db-5a1266f0324c","resolution":{"observed_at":"2026-08-04T18:51:20.997584Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.001407Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.001407Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:f9ee7dc65f5f05479e551ab81ae7ce53bf4d7949932c206a3c06e86d0387fac3","observation_id":"d496196d-2ed3-4674-8f4c-32f81c7874b0","resolution":{"observed_at":"2026-08-04T18:51:21.001407Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-08-04T18:51:21.004855Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.004855Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:7469c3623debf9c6a9a66dbac61e0a91597e69ad6e1db12dbf7208a5d9ce8f6f","observation_id":"7a4f428d-ff2b-48bb-8435-8c428573856f","resolution":{"observed_at":"2026-08-04T18:51:21.004855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.008503Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.008503Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:62fd9abcdef77faece73fb165592dbb16a37f97eb6453c7efa34a2cb66c78422","observation_id":"7b02ebf4-8848-4ee3-9bf3-4a6995c627a5","resolution":{"observed_at":"2026-08-04T18:51:21.008503Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.10117","last_updated":"2024-12-25T11:54:03Z","snapshot_observed_at":"2026-08-09T04:36:59.879757Z","submitted_at":"2024-12-13T12:59:39Z","title":"CosyVoice 2: Scalable Streaming Speech Synthesis with Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.10117","snapshot_observed_at":"2026-08-04T18:51:21.011679Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.011679Z"},"links":{"cited_paper":"/paper/2412.10117","citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:eb0da3228b5785c800acf6b7541158ea15f7ed2848382afb253f0b1a5199de7f","observation_id":"623cfdca-c120-4334-80d0-55ad82c8c7e1","resolution":{"observed_at":"2026-08-04T18:51:21.011679Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.015168Z","title":"E.; Wang, X.; Thakker, M.; Li, C.; Tsai, C.-H.; Xiao, Z.; Yang, H.; Zhu, Z.; Tang, M.; Tan, X.; et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.015168Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:3b88cb2afae1c6a3a56be640871e8b633649b19e0a3d201ea88307df960f603f","observation_id":"132ae60a-ebcb-4cad-8854-0f50b35a5bb0","resolution":{"observed_at":"2026-08-04T18:51:21.015168Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.11234","last_updated":"2025-03-12T16:27:37Z","snapshot_observed_at":"2026-08-07T18:14:25.986298Z","submitted_at":"2025-02-16T18:59:11Z","title":"MaskFlow: Discrete Flows For Flexible and Efficient Long Video Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.11234","snapshot_observed_at":"2026-08-04T18:51:21.018667Z","title":"T.; and Ommer, B","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.018667Z"},"links":{"cited_paper":"/paper/2502.11234","citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:30ab931a985693e608cb46119dff3c451b52671e95a9ecc04ddb60e5142eac67","observation_id":"449af1f8-014c-4dcc-89ce-f91297c5e11d","resolution":{"observed_at":"2026-08-04T18:51:21.018667Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.022179Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.022179Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:c8edfb93c0675ca31e693dd7e2cf7f1ddf0b71277361bed45dff5f3098763481","observation_id":"5f99fbfe-918f-4061-aa00-5180169529b1","resolution":{"observed_at":"2026-08-04T18:51:21.022179Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.025258Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.025258Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:3b237ad2ae2656085a0c52c586fa02c511e8a1948dfb2192acf139169b8e7fcf","observation_id":"0c8cd545-4b13-419b-9d61-bcbaf1d6f12d","resolution":{"observed_at":"2026-08-04T18:51:21.025258Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07855","last_updated":"2024-06-12T04:09:44Z","snapshot_observed_at":"2026-08-12T23:44:58.302920Z","submitted_at":"2024-06-12T04:09:44Z","title":"VALL-E R: Robust and Efficient Zero-Shot Text-to-Speech Synthesis via Monotonic Alignment","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.07855","snapshot_observed_at":"2026-08-04T18:51:21.028424Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.028424Z"},"links":{"cited_paper":"/paper/2406.07855","citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:d8067191c32756bec5696a59eda4a10b757ef4146ae20d2d07acd339cd284369","observation_id":"2e059189-d9c6-48b4-9101-5cd600e5de40","resolution":{"observed_at":"2026-08-04T18:51:21.028424Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.031774Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.031774Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:c0aa60b69dfbab29ed13b85b430e4b6583a8623924066c26fa37258eb1a3e1ff","observation_id":"6e54303f-7c14-4ae5-a0f2-7e51120bfc6e","resolution":{"observed_at":"2026-08-04T18:51:21.031774Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.034761Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.034761Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:3733943e79c4a9b4847f15917dcc722102c934294584747dd6dedc8024837eb7","observation_id":"4ac2d316-3d21-4e55-b251-0655cf177ba4","resolution":{"observed_at":"2026-08-04T18:51:21.034761Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.037693Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.037693Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:ae2cbc4669101db99a6cbbe690f340da6a4707919b4563152accb21b97c26e4d","observation_id":"bb24f6ad-d78e-4ee0-a1c6-1c8aa3984fef","resolution":{"observed_at":"2026-08-04T18:51:21.037693Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.040855Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.040855Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:560dbd9bd3fc336f3f32a56d91002cabfce8bed9c90246a683375aea95cc9c49","observation_id":"b9015e4f-d0fb-44c4-b6ca-e27548d083cd","resolution":{"observed_at":"2026-08-04T18:51:21.040855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.044039Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.044039Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:5aae29a540996f010612a1b668ddd3d2ac1091ad579e97429abb39f41c1f9488","observation_id":"a1bb253d-0b9a-4295-aa6d-40dd3db71a9e","resolution":{"observed_at":"2026-08-04T18:51:21.044039Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.047105Z","title":"J.; and Yang, E","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.047105Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:5695b95405381400b7bd0a82743a5c34d513ec1411660b25634115677c5569ba","observation_id":"f41de938-d538-4608-ae6f-0aa729cee034","resolution":{"observed_at":"2026-08-04T18:51:21.047105Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.050208Z","title":"J.; Badlani, R.; Santos, J","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.050208Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:51c9e38603545e04fa939e3397d171942de87c9f6c746de4af4b5bdc4eedace5","observation_id":"bed40b8f-cad7-4e43-820b-d5eeea232d1f","resolution":{"observed_at":"2026-08-04T18:51:21.050208Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.053137Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.053137Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:3911bfdb2bfa588728f809819605c0e54cb97aa6c0f6371f71cda675dc6d06c7","observation_id":"577f4734-562f-4b01-a383-c7fe9a4cbf7a","resolution":{"observed_at":"2026-08-04T18:51:21.053137Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.056152Z","title":"W.; Kim, J.; Chung, S.; and Cho, J","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.056152Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:7f94b90099ae9365592849a83b2cf66545502bb854cafd16523b9c22a136fef5","observation_id":"9fc02d3f-686a-4b27-a9c5-a9f6884e2bbc","resolution":{"observed_at":"2026-08-04T18:51:21.056152Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.059118Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.059118Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:e56168818e5292e4ec2d4045d0efdd3d7d9fe49ec62f7edc59d07ea43a655cfd","observation_id":"7882e6ab-bb84-4a83-8163-8731b69c3675","resolution":{"observed_at":"2026-08-04T18:51:21.059118Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.062079Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.062079Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:a47c778964a353cc5d15801d6dfa275885c39c25fb27ee38fc1bab49d3626bcd","observation_id":"13b654f6-10a2-4eb4-88cc-554c7694482c","resolution":{"observed_at":"2026-08-04T18:51:21.062079Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.064972Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.064972Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:e048e0bde1fc5964516c0cc4539c621182fa0fcd208c742cb153452d8ca6397a","observation_id":"cb92d948-25b9-4e11-8f3b-e91a53deb75b","resolution":{"observed_at":"2026-08-04T18:51:21.064972Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.069515Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.069515Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:ca53d87c1f14a59dbc4755c951414533fef2880c17edb57fa60b1d121662423c","observation_id":"c79a326d-f1c9-4250-8850-0a1fbfdadd98","resolution":{"observed_at":"2026-08-04T18:51:21.069515Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.072604Z","title":"M.; and Wei, F","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.072604Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:4107a946d3b44062806f5a52cae8edb64ad7c9949137e45923459a0bd100ebcc","observation_id":"42ab2445-945a-4f17-b31f-30bdbac225d5","resolution":{"observed_at":"2026-08-04T18:51:21.072604Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.075621Z","title":null,"venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.075621Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:59e377f00fac26efd566cb3bfecccaee05c069ea425d3dcd278bba997b8b4e6d","observation_id":"3147c038-1c93-4d2c-bdb3-239195e5f44c","resolution":{"observed_at":"2026-08-04T18:51:21.075621Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.078517Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.078517Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:55deb72c45c6301ef8c45db2febe00a593d95904c60226ba5245743f75f5785d","observation_id":"c6fb6efc-16bf-4d06-95b9-c7facaf1763b","resolution":{"observed_at":"2026-08-04T18:51:21.078517Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.081242Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.081242Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:66e917a77d60dabc71e2e460ba1e5f5b52a4d18c596363caea3bcaf3eb29d438","observation_id":"1e22373e-4d3b-4e48-b683-980b8d8336b5","resolution":{"observed_at":"2026-08-04T18:51:21.081242Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.084048Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.084048Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:e3ddd65bf78c78f99ad1177236191038c2ae9a269caa9f2d29a480147d60b37c","observation_id":"d62a951f-cced-4ac7-8c52-0ac2984c7baa","resolution":{"observed_at":"2026-08-04T18:51:21.084048Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.087056Z","title":"S.; Arriola, M.; Gokaslan, A.; Marroquin, E","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.087056Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:1e729f94f049f9d3d759117f7e5b4483637cc99808db187e9d27cf7938679363","observation_id":"7f8362ed-5e46-45a0-a273-921059ddbeb9","resolution":{"observed_at":"2026-08-04T18:51:21.087056Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.090018Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.090018Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:ea10ef92474a71bab34771fccc1e12258a856dc2e5cbc4f4fcb45c09a5d2d5b3","observation_id":"bd7304c7-58d2-4167-8f23-81baaed9ad42","resolution":{"observed_at":"2026-08-04T18:51:21.090018Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.093224Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.093224Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:ca86b053780a2f1d661725045b7129f6d2150907a127bc80e5ce7f7b14c4ee7e","observation_id":"b329725f-c0c1-4642-a276-940a4534ec97","resolution":{"observed_at":"2026-08-04T18:51:21.093224Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.096010Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.096010Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:6184978ca65f5e32fc3fd0a60730f8c77eeb713f75dfce94699769872600f6ba","observation_id":"bab2983b-1987-401f-af55-48830c156325","resolution":{"observed_at":"2026-08-04T18:51:21.096010Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07333","last_updated":"2024-01-14T17:43:55Z","snapshot_observed_at":"2026-08-12T19:56:04.603622Z","submitted_at":"2024-01-14T17:43:55Z","title":"ELLA-V: Stable Neural Codec Language Modeling with Alignment-guided Sequence Reordering","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.07333","snapshot_observed_at":"2026-08-04T18:51:21.098815Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.098815Z"},"links":{"cited_paper":"/paper/2401.07333","citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:1122a96efdfdfd537e8c6cca4aad68d891c12ce45ea553844607ea7ab8066f0c","observation_id":"d9ad5e0b-6bdc-447e-a44f-faaf8d83c312","resolution":{"observed_at":"2026-08-04T18:51:21.098815Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.102079Z","title":"P.; Kumar, A.; Ermon, S.; and Poole, B","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.102079Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:ba57368572233120a25fce3ffd0eb713ada4cd7374cb6f908eaf56e263f2ebef","observation_id":"130e0f9e-d5a0-4973-913a-665a5a8b2513","resolution":{"observed_at":"2026-08-04T18:51:21.102079Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.104955Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.104955Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:0092a87b7a39a0dd1226ccf6ecade94ccab3f821f2e2156e97baf86fe3f7033f","observation_id":"a50c2d85-a674-41dd-b5f6-3b1beffdd045","resolution":{"observed_at":"2026-08-04T18:51:21.104955Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.107937Z","title":"N.; Kaiser, L","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.107937Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:1ba69ba6e5a64263fca77f215efe4e1ada7f3ed3371d135f1b9dfc1eb316423c","observation_id":"55059669-68d8-4bed-9ac1-a4bf5924e5f1","resolution":{"observed_at":"2026-08-04T18:51:21.107937Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.24291","last_updated":"2025-05-30T07:04:23Z","snapshot_observed_at":"2026-08-07T12:24:06.199283Z","submitted_at":"2025-05-30T07:04:23Z","title":"Discl-VC: Disentangled Discrete Tokens and In-Context Learning for Controllable Zero-Shot Voice Conversion","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.24291","snapshot_observed_at":"2026-08-04T18:51:21.110818Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.110818Z"},"links":{"cited_paper":"/paper/2505.24291","citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:93455fa28c38e4d6af1ba1fa9252a45507a108088e80d6466e5ea9c87d333486","observation_id":"e305150d-f29c-4ca9-b00b-0b8561adc764","resolution":{"observed_at":"2026-08-04T18:51:21.110818Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01710","last_updated":"2025-03-03T16:23:10Z","snapshot_observed_at":"2026-07-06T20:45:50.730195Z","submitted_at":"2025-03-03T16:23:10Z","title":"Spark-TTS: An Efficient LLM-Based Text-to-Speech Model with Single-Stream Decoupled Speech Tokens","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01710","snapshot_observed_at":"2026-08-04T18:51:21.114170Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.114170Z"},"links":{"cited_paper":"/paper/2503.01710","citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:d148df61b53aecf5418645f91e234112f54dd7314bfae7d9f14ce6c1b7c117d9","observation_id":"af68322a-adf3-429a-9b59-fdc25d09a134","resolution":{"observed_at":"2026-08-04T18:51:21.114170Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.117594Z","title":"J.; and Liao, R","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.117594Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:6aeab0b573c930eaa4f93409bcb4ec68ee769c13f8412d64df86b2a57d93d62f","observation_id":"88eb04bc-47ea-4b84-90c4-2f1e538e8a22","resolution":{"observed_at":"2026-08-04T18:51:21.117594Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.120760Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.120760Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:ba5e93aa891993ef36953063c0c562f97ae2f3064f1e580cdf66fad21de4725c","observation_id":"f504b305-d9ae-4b7a-beea-b107ba01299e","resolution":{"observed_at":"2026-08-04T18:51:21.120760Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.124066Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.124066Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:41f3b6a4127d1c54293245387f7da02fcc8b69da0ecb06d2c3ff8d1ec56fcf73","observation_id":"37eed620-c63b-4701-b804-7df70c12f746","resolution":{"observed_at":"2026-08-04T18:51:21.124066Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.14928","last_updated":"2025-03-19T06:28:17Z","snapshot_observed_at":"2026-08-09T10:39:29.929697Z","submitted_at":"2025-03-19T06:28:17Z","title":"Shushing! Let's Imagine an Authentic Speech from the Silent Video","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.14928","snapshot_observed_at":"2026-08-04T18:51:21.127479Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.127479Z"},"links":{"cited_paper":"/paper/2503.14928","citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:f0a46b187d9f967fb716173b4a32618943955365cfee761113832909c8e68cf6","observation_id":"fc9a04e0-63b2-4603-a053-9aca3376b72b","resolution":{"observed_at":"2026-08-04T18:51:21.127479Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.130764Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.130764Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:d307a2cc06a5543ed176802e49be0fae8ade98d8a092c71052eeacd25e22ec5c","observation_id":"1aaa4b9a-6bd8-43d2-be64-46877872e18b","resolution":{"observed_at":"2026-08-04T18:51:21.130764Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.133684Z","title":"J.; Jia, Y.; Chen, Z.; and Wu, Y","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.133684Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:b99f8509dedf5935d336d344aaea6e3074b513404b58d32cb34ad5474bf55b82","observation_id":"d00bd9a2-5f9e-4a02-979e-fbeaf38fb27e","resolution":{"observed_at":"2026-08-04T18:51:21.133684Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.03926","last_updated":"2023-03-07T14:31:55Z","snapshot_observed_at":"2026-08-06T04:55:08.186019Z","submitted_at":"2023-03-07T14:31:55Z","title":"Speak Foreign Languages with Your Own Voice: Cross-Lingual Neural Codec Language Modeling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.03926","snapshot_observed_at":"2026-08-04T18:51:21.136655Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.136655Z"},"links":{"cited_paper":"/paper/2303.03926","citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:955977bf27be3498bb8a1effe70bbcc5b883415f799a4c7c8a50cb9a1bb986dd","observation_id":"9dc06219-a9e9-4be9-b4e1-e28e44f6aed5","resolution":{"observed_at":"2026-08-04T18:51:21.136655Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.139897Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.139897Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:77518fb2d9b0671231e13a7c18f515cddeb4a15bc133ef291c29395c4d5adc88","observation_id":"64b4c90c-0d4e-4fe9-8b47-db9b50b22fa8","resolution":{"observed_at":"2026-08-04T18:51:21.139897Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T18:51:21.143066Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.143066Z"},"links":{"citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:979a4a5425b04269496175b6d4f0d70786c03dd6757f0f70ddff03978132fbce","observation_id":"f62205af-9979-4389-8464-e43fcf10657a","resolution":{"observed_at":"2026-08-04T18:51:21.143066Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","latest_version":5,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-08T11:03:42.227884Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching"},"reference_resolution":{"displayed":52,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":52,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":52},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 52 of 52 outbound references and 1 inbound Pith citation observation for arXiv:2509.09631."}