{"as_of":"2026-08-09T07:27:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:8dc0b1e9fe4de6101397a2c16b4deb46d3e2395e7b72baa319697a0f4b33b36b","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":36,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":36,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":36,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":36,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T05:54:18.726951Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T18:40:03.192889Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":139,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:a3cc7dd46e92f8ca5f980bf0eb8a42231123ae50a3c29646e82c0b8219e97857","observation_id":"f47485fe-6dab-43e4-b8fe-365529144a4b","resolution":{"observed_at":"2026-05-16T06:06:41.517814Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2410.13720","last_updated":"2025-02-26T16:05:55Z","snapshot_observed_at":"2026-08-09T03:17:30.671491Z","submitted_at":"2024-10-17T16:22:46Z","title":"Movie Gen: A Cast of Media Foundation Models","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-11T14:16:18.521699Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2410.13720"},"observation_digest":"sha256:7a5a6bb616dd20efcf09af9a968bff94546b3d8d89d7d2c431a3489070ae0539","observation_id":"0c322811-7749-48e4-bc4b-4936f82b6af3","resolution":{"observed_at":"2026-05-11T14:16:25.914341Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-09T05:54:18.726951Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.03128","last_updated":"2025-02-05T12:36:21Z","snapshot_observed_at":"2026-08-09T05:47:19.180290Z","submitted_at":"2025-02-05T12:36:21Z","title":"Metis: A Foundation Speech Generation Model with Masked Generative Pre-training","version":1},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-08-09T05:54:18.726951Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2502.03128"},"observation_digest":"sha256:cfad85f936334d00f2bc3c4b76fba258b3e342874b062eb0e1a2f10e4f710ea4","observation_id":"6315151a-bfd4-449a-b68c-b0427656b917","resolution":{"observed_at":"2026-08-09T05:54:18.726951Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-07T14:19:56.312162Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers.ArXiv, abs/2304.09116, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19462","last_updated":"2025-05-31T22:36:04Z","snapshot_observed_at":"2026-08-09T03:57:33.373890Z","submitted_at":"2025-05-26T03:35:44Z","title":"VoiceStar: Robust Zero-Shot Autoregressive TTS with Duration Control and Extrapolation","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:56.312162Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2505.19462"},"observation_digest":"sha256:603993f08253db9969cb70f7be655d361386f099ad9c60964c56fc3a54042455","observation_id":"d5407115-8935-4156-98a6-2901d3393e33","resolution":{"observed_at":"2026-08-07T14:19:56.312162Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-07T12:12:56.445581Z","title":"Naturalspeech 2: Latent diffusion models are natu- ral and zero-shot speech and singing synthesizers,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.00350","last_updated":"2025-05-31T02:23:38Z","snapshot_observed_at":"2026-08-07T23:23:36.425454Z","submitted_at":"2025-05-31T02:23:38Z","title":"DiffDSR: Dysarthric Speech Reconstruction Using Latent Diffusion Model","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T12:12:56.445581Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2506.00350"},"observation_digest":"sha256:5d4030ea8853eeabe086b3fd5f9d47dd05c765acdc2154406340487ab9a6e468","observation_id":"74c40840-3b67-4339-a24d-74fc4cd50c58","resolution":{"observed_at":"2026-08-07T12:12:56.445581Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-07T12:03:37.184518Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.00868","last_updated":"2025-06-16T12:09:09Z","snapshot_observed_at":"2026-08-09T04:53:21.850036Z","submitted_at":"2025-06-01T07:17:16Z","title":"Multiverse Through Deepfakes: The MultiFakeVerse Dataset of Person-Centric Visual and Conceptual Manipulations","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:37.184518Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2506.00868"},"observation_digest":"sha256:a09ce4464a3589544e04694611491dfcbf71a65153b6f84ac439fa54f1c2b9a1","observation_id":"84115302-5a8a-4079-8810-af0dbd16bb2d","resolution":{"observed_at":"2026-08-07T12:03:37.184518Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2506.23552","last_updated":"2026-05-15T04:47:28Z","snapshot_observed_at":"2026-07-06T21:49:33.849898Z","submitted_at":"2025-06-30T06:51:40Z","title":"JAM-Flow: Joint Audio-Motion Synthesis with Flow Matching","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-22T00:46:39.196042Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2506.23552"},"observation_digest":"sha256:53499780008a47dc9ce920ed37fb70cb7857d83e927d33aaf186858e8d585aa3","observation_id":"4e224cf6-1374-4e0e-851a-8f60fdf62d64","resolution":{"observed_at":"2026-05-22T00:50:51.041243Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-06T21:00:07.579047Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.01348","last_updated":"2025-07-08T09:21:24Z","snapshot_observed_at":"2026-08-08T00:30:57.740737Z","submitted_at":"2025-07-02T04:30:23Z","title":"SpeechAccentLLM: A Unified Framework for Foreign Accent Conversion and Text to Speech","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-06T21:00:07.579047Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2507.01348"},"observation_digest":"sha256:e1ff8002bc695e923747d51a099fd1e8885677c0ea20f6b67ccd3eab9930bce6","observation_id":"ec02f16b-1a6e-4d7c-89f3-9d90ad4f2018","resolution":{"observed_at":"2026-08-06T21:00:07.579047Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-06T15:48:21.184511Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.14988","last_updated":"2025-07-20T14:48:48Z","snapshot_observed_at":"2026-08-08T10:46:13.682261Z","submitted_at":"2025-07-20T14:48:48Z","title":"DMOSpeech 2: Reinforcement Learning for Duration Prediction in Metric-Optimized Speech Synthesis","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:21.184511Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2507.14988"},"observation_digest":"sha256:25cacd3e6f51935ea0a56664efdf9418ffa8bbc25aea3645f39f49e85eff1c02","observation_id":"a96b3cfc-3d52-47d9-b621-0ef2835b38f1","resolution":{"observed_at":"2026-08-06T15:48:21.184511Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-06T14:04:00.825953Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.19835","last_updated":"2025-07-26T07:12:02Z","snapshot_observed_at":"2026-08-07T16:55:18.620169Z","submitted_at":"2025-07-26T07:12:02Z","title":"SonicGauss: Position-Aware Physical Sound Synthesis for 3D Gaussian Representations","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T14:04:00.825953Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2507.19835"},"observation_digest":"sha256:9a8f234a7d077b4840b37bd314910a57f3279cd470fe0e757157c82f357fab31","observation_id":"3b20f27b-795a-4b6b-b9ef-dba358d4ba82","resolution":{"observed_at":"2026-08-06T14:04:00.825953Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-06T11:22:27.121148Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.22746","last_updated":"2025-08-01T03:37:42Z","snapshot_observed_at":"2026-08-06T13:36:59.117683Z","submitted_at":"2025-07-30T15:03:36Z","title":"Next Tokens Denoising for Speech Synthesis","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T11:22:27.121148Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2507.22746"},"observation_digest":"sha256:1426a2893eae456d446c196660610bcafc63076f37ab2e36c295f2f187dd2853","observation_id":"faba2588-9e08-487e-b264-81415b3455e9","resolution":{"observed_at":"2026-08-06T11:22:27.121148Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-06T05:17:06.401559Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.02038","last_updated":"2025-08-14T01:47:28Z","snapshot_observed_at":"2026-08-08T00:25:19.301459Z","submitted_at":"2025-08-04T04:08:22Z","title":"Marco-Voice Technical Report","version":4},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T05:17:06.401559Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2508.02038"},"observation_digest":"sha256:6fd7052cdc93378ec0d79b5f8ec2bad4ac9cc4d152520621e76e5438dfe204de","observation_id":"3dd2c9a9-4ed4-4bde-be05-e8b4c8dbfb64","resolution":{"observed_at":"2026-08-06T05:17:06.401559Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-06T05:05:28.156213Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers.arXiv preprint arXiv:2304.09116, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.02391","last_updated":"2025-08-04T13:17:49Z","snapshot_observed_at":"2026-08-08T02:50:04.116531Z","submitted_at":"2025-08-04T13:17:49Z","title":"Inference-time Scaling for Diffusion-based Audio Super-resolution","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T05:05:28.156213Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2508.02391"},"observation_digest":"sha256:e69cdc82aaeb1322c578cc36157668c189ef776592297b9ef719f6c1afd9ca64","observation_id":"e8139e27-ff8e-4e4d-86e3-ba24bdd26ffb","resolution":{"observed_at":"2026-08-06T05:05:28.156213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-05T15:10:31.804516Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.20379","last_updated":"2025-08-28T03:00:30Z","snapshot_observed_at":"2026-08-08T13:26:19.436875Z","submitted_at":"2025-08-28T03:00:30Z","title":"Audio-Guided Visual Editing with Complex Multi-Modal Prompts","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-05T15:10:31.804516Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2508.20379"},"observation_digest":"sha256:e9ed09f979954ff9a0d6e743532a58e36968f564417e095a02a4b342479f9f4e","observation_id":"6562b558-0a3e-403b-9f40-55481a3dceab","resolution":{"observed_at":"2026-08-05T15:10:31.804516Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-05T13:35:01.874077Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero- shot speech and singing synthesizers,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-08T00:31:06.181577Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:01.874077Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:45da6947aa885961638f5807a14f945db25b91d359ce3d281f8b0496ad2d22f6","observation_id":"74f964eb-a931-415f-8f63-f15b96a40155","resolution":{"observed_at":"2026-08-05T13:35:01.874077Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2509.18060","last_updated":"2026-04-18T03:04:41Z","snapshot_observed_at":"2026-08-08T00:01:52.882877Z","submitted_at":"2025-09-22T17:38:52Z","title":"TMD-TTS: A Unified Tibetan Multi-Dialect Text-to-Speech Framework for \\\"U-Tsang, Amdo and Kham Speech Dataset Generation","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-18T14:25:40.217180Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2509.18060"},"observation_digest":"sha256:33e3257c54aeaac4c8cdbd5d581fb4893ba5e55a81783ed385cd0f5238d8933d","observation_id":"2ecbb115-ba05-42da-bc95-60a3ee269242","resolution":{"observed_at":"2026-05-18T14:26:27.981615Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-04T11:29:35.214098Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2510.04593","last_updated":"2026-06-08T05:49:30Z","snapshot_observed_at":"2026-08-06T08:55:28.794308Z","submitted_at":"2025-10-06T08:47:38Z","title":"UniVoice: Unifying Autoregressive ASR and Flow-Matching based TTS with Large Language Models","version":3},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-04T11:29:35.214098Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2510.04593"},"observation_digest":"sha256:7d8dcc229c281b9d8a2c942fae707150c541c22b5e290ab0782929b9040f18cc","observation_id":"5aa886f0-7a3f-46af-a0b0-bdfb212737bb","resolution":{"observed_at":"2026-08-04T11:29:35.214098Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2601.15621","last_updated":"2026-01-22T03:51:43Z","snapshot_observed_at":"2026-08-04T12:22:34.670840Z","submitted_at":"2026-01-22T03:51:43Z","title":"Qwen3-TTS Technical Report","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-16T19:24:56.057631Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2601.15621"},"observation_digest":"sha256:e94f316b8631d5f91036727f100e930882e7b9aba0a16ed55a85de335d24e770","observation_id":"2953e945-6422-45e4-b6ab-0d92a97d0c64","resolution":{"observed_at":"2026-05-16T19:24:56.145264Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-03T00:11:16.339880Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers.arXiv preprint arXiv:2304.09116,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.12304","last_updated":"2026-07-23T11:24:08Z","snapshot_observed_at":"2026-08-03T00:11:08.466564Z","submitted_at":"2026-02-12T03:25:41Z","title":"OmniCustom: Sync Audio-Video Customization Via Joint Audio-Video Generation Model","version":5},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-03T00:11:16.339880Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2602.12304"},"observation_digest":"sha256:fb2ba1f967cd40ed9c0c9f8bcd70e4cb2004da59cbcf063a3e4c52d21fe9c3f5","observation_id":"aa40bc84-52a8-4ae8-914d-109341bc5d0d","resolution":{"observed_at":"2026-08-03T00:11:16.339880Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2604.06327","last_updated":"2026-04-07T18:05:28Z","snapshot_observed_at":"2026-07-06T22:54:48.916799Z","submitted_at":"2026-04-07T18:05:28Z","title":"A Novel Automatic Framework for Speaker Drift Detection in Synthesized Speech","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T18:24:54.627215Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2604.06327"},"observation_digest":"sha256:f2f84f62c6b9d2bf1411d34afee1106c7b44a544ac4a0bb5c5bd6762cba95f5a","observation_id":"b4843fa1-2926-4e71-8322-db573e7e4dc3","resolution":{"observed_at":"2026-05-11T00:41:00.491000Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2604.11283","last_updated":"2026-06-01T08:50:38Z","snapshot_observed_at":"2026-08-02T05:26:21.584719Z","submitted_at":"2026-04-13T10:42:31Z","title":"Multimodal Large Language Model-Enabled Video Translation: A Role-Oriented Survey","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-05-10T16:36:33.264166Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2604.11283"},"observation_digest":"sha256:62f04e2220e64f495034ef8db3b16214c6a09863302d0efb96df84a2e390b159","observation_id":"5e7439ab-4d74-46fe-b29a-8072b6e7f30e","resolution":{"observed_at":"2026-05-11T08:30:57.199763Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-12T22:04:31.302192Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2604.11283","last_updated":"2026-06-01T08:50:38Z","snapshot_observed_at":"2026-08-02T05:26:21.584719Z","submitted_at":"2026-04-13T10:42:31Z","title":"Multimodal Large Language Model-Enabled Video Translation: A Role-Oriented Survey","version":2},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-07-12T22:04:31.302192Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2604.11283"},"observation_digest":"sha256:a54845d29fb6d316646197faf0b5d54c52fa98584b2177e449b920cabcf1e999","observation_id":"46c4a353-49c8-4094-9e8c-fc49dce02c09","resolution":{"observed_at":"2026-07-12T22:04:31.302192Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2604.24416","last_updated":"2026-04-27T12:45:18Z","snapshot_observed_at":"2026-07-06T23:10:29.508602Z","submitted_at":"2026-04-27T12:45:18Z","title":"Scaling Properties of Continuous Diffusion Spoken Language Models","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-08T03:43:10.132447Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2604.24416"},"observation_digest":"sha256:869dcd9cf9a58095a271fe478265d20ebbb5240d0332af0a33c2ec6e0474e94d","observation_id":"2c603281-0c77-48ca-a921-0538cb1e2ce2","resolution":{"observed_at":"2026-05-11T21:56:28.356045Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2605.16578","last_updated":"2026-05-26T19:32:15Z","snapshot_observed_at":"2026-08-07T18:10:01.317172Z","submitted_at":"2026-05-15T19:32:28Z","title":"Voice \"Cloning\" is Style Transfer","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-30T18:58:45.582395Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2605.16578"},"observation_digest":"sha256:672a13e15527a1bae42e49ae6409d0c6b598fc528c4848909d0bb6ef7935286b","observation_id":"648a05cf-8f2f-41f5-a8bc-d95b2bd7ed38","resolution":{"observed_at":"2026-06-30T19:05:01.038014Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2605.16964","last_updated":"2026-05-16T12:37:06Z","snapshot_observed_at":"2026-08-02T05:10:05.598125Z","submitted_at":"2026-05-16T12:37:06Z","title":"SemaVoice: Semantic-Aware Continuous Autoregressive Speech Synthesis","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-19T18:58:15.288299Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2605.16964"},"observation_digest":"sha256:fb0d425d531253991e5ae815f783d2ac26211be029db8ee186126b5ce162ba08","observation_id":"b5a2da5f-67c4-4826-b9c7-27bf2f535ccc","resolution":{"observed_at":"2026-05-19T19:02:43.563073Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2606.05852","last_updated":"2026-06-04T08:27:17Z","snapshot_observed_at":"2026-08-04T06:57:18.394840Z","submitted_at":"2026-06-04T08:27:17Z","title":"UniVoice: A Unified Model for Speech and Singing Voice Generation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-27T23:56:14.198308Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2606.05852"},"observation_digest":"sha256:190462e3af4675ab88fac60e9c2de198a38945fc00efc80fc6a487afaf56afcb","observation_id":"f93442d7-5bf7-464a-b5f6-aadf112dbf63","resolution":{"observed_at":"2026-07-02T15:17:08.298843Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2606.20650","last_updated":"2026-06-08T06:54:55Z","snapshot_observed_at":"2026-08-02T23:08:38.292740Z","submitted_at":"2026-06-08T06:54:55Z","title":"EmoInstruct-TTS: Dual-Path Instruction-Guided Emotional Speech Synthesis","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-27T17:01:13.972071Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2606.20650"},"observation_digest":"sha256:6805782fadea70514002bc86967007601490e272789566fff4a24365babadc96","observation_id":"0b70f74c-f86a-40bd-aac6-a1530e6b36e5","resolution":{"observed_at":"2026-07-03T00:47:30.531681Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2606.23489","last_updated":"2026-06-22T15:35:30Z","snapshot_observed_at":"2026-07-06T23:58:11.694332Z","submitted_at":"2026-06-22T15:35:30Z","title":"MeshFlow: Mesh Generation with Equivariant Flow Matching","version":1},"reference_index":103,"source":"arxiv_source","source_observed_at":"2026-06-26T05:52:47.536542Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2606.23489"},"observation_digest":"sha256:318f9d594f9f1d4ff941cec531a565077bd251eb5eb01750e208bbe85136a1a9","observation_id":"99b1b84c-9075-47ee-a0fb-c22b443a5885","resolution":{"observed_at":"2026-07-04T12:49:52.683546Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2606.24320","last_updated":"2026-06-25T20:48:22Z","snapshot_observed_at":"2026-07-06T23:58:54.018557Z","submitted_at":"2026-06-23T08:57:34Z","title":"ZONOS2 Technical Report","version":1},"reference_index":190,"source":"arxiv_source","source_observed_at":"2026-06-25T22:37:15.072758Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2606.24320"},"observation_digest":"sha256:b5f87c12fc4ee3a9a3b96ed32020aa24dcbf7ceb48624b092d82df0dcbc3ef44","observation_id":"ef844789-06ca-4d27-aa25-2caf8243ba5c","resolution":{"observed_at":"2026-07-04T18:40:03.194270Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2606.24320","last_updated":"2026-06-25T20:48:22Z","snapshot_observed_at":"2026-07-06T23:58:54.018557Z","submitted_at":"2026-06-23T08:57:34Z","title":"ZONOS2 Technical Report","version":2},"reference_index":190,"source":"arxiv_source","source_observed_at":"2026-06-29T02:07:31.791835Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2606.24320"},"observation_digest":"sha256:23de52507492a7da9dbc4c8413b88e1d89e8dfcf6fe24ba80b5a0bd40d418fa0","observation_id":"6f2f3709-6fd6-46c7-a37b-afc570b0240e","resolution":{"observed_at":"2026-07-01T18:15:59.065739Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2606.31247","last_updated":"2026-06-30T07:24:10Z","snapshot_observed_at":"2026-07-07T00:04:58.486133Z","submitted_at":"2026-06-30T07:24:10Z","title":"FlexiSLM: A Dynamic and Controllable Frame Rate Spoken Language Model","version":1},"reference_index":150,"source":"arxiv_source","source_observed_at":"2026-07-01T03:50:26.873406Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2606.31247"},"observation_digest":"sha256:b75eda01da69e5e769432698202d87bd34f20c0e68e40ef0a1d295157a972ff0","observation_id":"78488924-78b0-442e-9b36-f069452bdea2","resolution":{"observed_at":"2026-07-01T11:45:47.268582Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-01T11:43:06.058605Z","title":"arXiv preprint arXiv:2304.09116 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.19810","last_updated":"2026-07-22T06:42:20Z","snapshot_observed_at":"2026-08-07T23:49:52.327639Z","submitted_at":"2026-07-22T06:42:20Z","title":"SimulS2ST-Omni: Data-Efficient Streaming Speech-to-Speech Translation via Explicit Trajectory Supervision","version":1},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-08-01T11:43:06.058605Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2607.19810"},"observation_digest":"sha256:96a51d94e34c7e9e1f474454bac37b5639d4985e99158b028f01ee48b07afbd4","observation_id":"f91965fd-1d8d-4831-aa05-a7edb4019bbb","resolution":{"observed_at":"2026-08-01T11:43:06.058605Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-01T11:33:12.827054Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.19859","last_updated":"2026-07-22T07:48:37Z","snapshot_observed_at":"2026-08-07T12:49:18.178302Z","submitted_at":"2026-07-22T07:48:37Z","title":"StellarTTS: Sparse Temporal Embedding for Low-Latency and Robust Speech Synthesis","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:12.827054Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2607.19859"},"observation_digest":"sha256:b6ff3b1722859a47569ffe53b2319500088c6b97c7bae67d83d0d8f6abc7596e","observation_id":"feb60fb7-d20c-49c9-93c2-cc2bafa5bac3","resolution":{"observed_at":"2026-08-01T11:33:12.827054Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-06T00:40:31.445968Z","title":"Naturalspeech 2: Latent diffusion models are natu- ral and zero-shot speech and singing synthesizers,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.00998","last_updated":"2026-08-02T04:54:12Z","snapshot_observed_at":"2026-08-08T21:36:52.375893Z","submitted_at":"2026-08-02T04:54:12Z","title":"Beyond One-Size-Fits-All: Personalized and Culturally Adaptive Emotional TTS via Interactive Optimization of Individual Emotion Perception Spaces","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T00:40:31.445968Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2608.00998"},"observation_digest":"sha256:696a8eb3f55750eda193764a89fe9060d1684d4f7db1a9d94cacd2f651b664a5","observation_id":"43fc6afc-8c13-4382-8c7d-7ab8c9e49b12","resolution":{"observed_at":"2026-08-06T00:40:31.445968Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-04T16:29:27.768440Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers.arXiv preprint arXiv:2304.09116, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.02023","last_updated":"2026-08-04T08:54:48Z","snapshot_observed_at":"2026-08-07T23:09:45.300110Z","submitted_at":"2026-08-03T10:20:52Z","title":"SwanTale: Unified Multi-Speaker Speech and Audio Generation for Instruct and Zero-Shot Tasks","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-04T16:29:27.768440Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2608.02023"},"observation_digest":"sha256:736e823feb80a40587839abbb277b5d7d28efdb220aefee8083b89bad7871445","observation_id":"21be1899-e25c-4269-8f96-a72d1861f903","resolution":{"observed_at":"2026-08-04T16:29:27.768440Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-07T00:14:47.793573Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers.arXiv preprint arXiv:2304.09116, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.02023","last_updated":"2026-08-04T08:54:48Z","snapshot_observed_at":"2026-08-07T23:09:45.300110Z","submitted_at":"2026-08-03T10:20:52Z","title":"SwanTale: Unified Multi-Speaker Speech and Audio Generation for Instruct and Zero-Shot Tasks","version":2},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-07T00:14:47.793573Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2608.02023"},"observation_digest":"sha256:0333efb9676cd4829d71a21e4b0397004e6a5cafe0f9328d026087f2c30d5fd4","observation_id":"265648e2-a0c1-4985-8af3-e4ba1b4ce885","resolution":{"observed_at":"2026-08-07T00:14:47.793573Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2304.09116/citation-record","integrity":"/paper/2304.09116/integrity","json":"/paper/2304.09116/citation-record.json","paper":"/paper/2304.09116"},"outbound":[],"paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","latest_version":3,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 36 inbound Pith citation observations for arXiv:2304.09116."}