{"as_of":"2026-08-23T04:58:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:99a1d9b35548685cee7eeda15ce54bc49db9bf3a9448fa2e52a59d45b02c34cd","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":40,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":40,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-22T06:32:14.747728+00:00","state":"measured"},{"denominator":40,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":40,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T00:37:04.982405Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":11,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":"2306.00814","doi":"10.48550/arxiv.2306.00814","metadata_source":"pith","pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis","venue":"cs.SD","work_id":"dfb28242-44a0-4f86-bb57-65dd1aae8947","year":2023},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-16T05:58:47.061997Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":140,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:b075b134f67512ef8e2a8eee9dc1c8f5ac0d68224eceeebf9d40a12dc493371f","observation_id":"a45dbe55-9889-4076-9708-b7e7239e9345","resolution":{"observed_at":"2026-05-16T06:06:41.521748Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-12T18:47:51.609553Z","title":"Vocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis.arXiv preprint arXiv:2306.00814, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.11258","last_updated":"2024-11-18T03:22:34Z","snapshot_observed_at":"2026-08-18T03:16:17.864870Z","submitted_at":"2024-11-18T03:22:34Z","title":"ESTVocoder: An Excitation-Spectral-Transformed Neural Vocoder Conditioned on Mel Spectrogram","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-12T18:47:51.609553Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2411.11258"},"observation_digest":"sha256:081738fe9dd344ce2d2a5f0ad8259cc07de3ceb0c6edfd945d8ef70ec308f3af","observation_id":"223fd8ff-ccae-411d-ad3f-1723c05de0e4","resolution":{"observed_at":"2026-08-12T18:47:51.609553Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-12T17:47:13.037199Z","title":"arXiv preprint arXiv:2306.00814 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.12268","last_updated":"2024-11-19T06:40:01Z","snapshot_observed_at":"2026-08-18T17:51:42.800664Z","submitted_at":"2024-11-19T06:40:01Z","title":"A Neural Denoising Vocoder for Clean Waveform Generation from Noisy Mel-Spectrogram based on Amplitude and Phase Predictions","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T17:47:13.037199Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2411.12268"},"observation_digest":"sha256:f399b47ea5f906c23b44f1ced45765ab88901bba6e3d4f822c179c041239b94e","observation_id":"ad3737eb-3bb6-48d8-bc61-6ccb7e8432de","resolution":{"observed_at":"2026-08-12T17:47:13.037199Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-12T10:57:08.526943Z","title":"V ocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.18803","last_updated":"2025-04-27T16:37:28Z","snapshot_observed_at":"2026-08-19T16:49:09.731359Z","submitted_at":"2024-11-27T23:07:52Z","title":"TS3-Codec: Transformer-Based Simple Streaming Single Codec","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T10:57:08.526943Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2411.18803"},"observation_digest":"sha256:b226ad984c7bddc81c0488fe5f993caf6fbca52fbf0e8a40d260bf834d09533f","observation_id":"82e50c85-551c-4381-b35f-509eeb55e5c5","resolution":{"observed_at":"2026-08-12T10:57:08.526943Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-09T10:37:03.701969Z","title":"V ocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2502.02942","last_updated":"2025-02-05T07:14:39Z","snapshot_observed_at":"2026-08-22T18:11:38.407455Z","submitted_at":"2025-02-05T07:14:39Z","title":"GenSE: Generative Speech Enhancement via Language Models using Hierarchical Modeling","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-09T10:37:03.701969Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2502.02942"},"observation_digest":"sha256:e94d89a575cdc86bdda275966e218c44e2f84c07535654ca041aedab0d7bfb9e","observation_id":"acc77a20-f55a-4746-86b1-74de8fce3ff3","resolution":{"observed_at":"2026-08-09T10:37:03.701969Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-09T05:54:18.731349Z","title":"Vocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.03128","last_updated":"2025-02-05T12:36:21Z","snapshot_observed_at":"2026-08-18T17:13:14.270187Z","submitted_at":"2025-02-05T12:36:21Z","title":"Metis: A Foundation Speech Generation Model with Masked Generative Pre-training","version":1},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-08-09T05:54:18.731349Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2502.03128"},"observation_digest":"sha256:ce2d2bcea9608e2aa8bcc7578a0f803a227bee7515648fc45419150a6dfa81c8","observation_id":"7d5ca28d-17ae-426b-a5b3-3936b1de4e81","resolution":{"observed_at":"2026-08-09T05:54:18.731349Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-15T20:30:50.836022Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.12800","last_updated":"2025-05-19T07:31:55Z","snapshot_observed_at":"2026-08-17T12:38:34.601646Z","submitted_at":"2025-05-19T07:31:55Z","title":"OZSpeech: One-step Zero-shot Speech Synthesis with Learned-Prior-Conditioned Flow Matching","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-15T20:30:50.836022Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2505.12800"},"observation_digest":"sha256:e14f2e7e45f613e2b9aaeba6a86b63561b086e5d2129e215e768123fd3440f81","observation_id":"1098c571-dcdf-4bfb-a94c-32ebe44b8a97","resolution":{"observed_at":"2026-08-15T20:30:50.836022Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-07T15:37:09.616746Z","title":"V ocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.14465","last_updated":"2025-05-20T15:01:30Z","snapshot_observed_at":"2026-08-22T04:09:15.665885Z","submitted_at":"2025-05-20T15:01:30Z","title":"FlowTSE: Target Speaker Extraction with Flow Matching","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T15:37:09.616746Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2505.14465"},"observation_digest":"sha256:9ff20d341cd74a3885c23cdb37316aebe92f820c4fadccb9546de205677c1522","observation_id":"e3045518-e5d9-4e24-b1e7-6bfbecd78b09","resolution":{"observed_at":"2026-08-07T15:37:09.616746Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-07T14:55:57.198647Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-20T21:09:56.333663Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:57.198647Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:887a3c341660a8246fe7cd9459b6c1399e432365531693858537a98fe5b63e34","observation_id":"fa95a800-6595-417b-af9a-f94e5bea44ef","resolution":{"observed_at":"2026-08-07T14:55:57.198647Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-07T14:18:26.198033Z","title":"V ocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19476","last_updated":"2025-05-27T09:33:56Z","snapshot_observed_at":"2026-08-16T06:01:00.086804Z","submitted_at":"2025-05-26T03:55:00Z","title":"FlowSE: Efficient and High-Quality Speech Enhancement via Flow Matching","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T14:18:26.198033Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2505.19476"},"observation_digest":"sha256:608849c6c7e87fbf35807edfa525106b7164cb60c7069be457406cecad63d450","observation_id":"d35f41a3-464e-4842-b976-c54aa9b8cf8e","resolution":{"observed_at":"2026-08-07T14:18:26.198033Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-07T14:15:18.545648Z","title":"V ocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19576","last_updated":"2025-05-26T06:47:30Z","snapshot_observed_at":"2026-08-18T17:54:25.751059Z","submitted_at":"2025-05-26T06:47:30Z","title":"Mel-McNet: A Mel-Scale Framework for Online Multichannel Speech Enhancement","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:18.545648Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2505.19576"},"observation_digest":"sha256:a6fff5500830861f350800fa084d12bce38df107fe1bd07835b3ce39336c5dd7","observation_id":"5a4498b8-5f69-4669-bc9a-9b2c2b66c83d","resolution":{"observed_at":"2026-08-07T14:15:18.545648Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-07T12:29:42.327056Z","title":"V ocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.24314","last_updated":"2025-05-30T07:53:01Z","snapshot_observed_at":"2026-08-14T10:04:40.808126Z","submitted_at":"2025-05-30T07:53:01Z","title":"DS-Codec: Dual-Stage Training with Mirror-to-NonMirror Architecture Switching for Speech Codec","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T12:29:42.327056Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2505.24314"},"observation_digest":"sha256:cd1e40d9691b5a24f8ed548c82f2b87283a45e2c566ea89f22175fbe2fdde9ce","observation_id":"f7213156-5f1e-49bc-937f-5e62869b6ad6","resolution":{"observed_at":"2026-08-07T12:29:42.327056Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-07T12:12:29.406719Z","title":"V ocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis.arXiv preprint arXiv:2306.00814,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.00385","last_updated":"2025-05-31T04:31:02Z","snapshot_observed_at":"2026-08-12T05:59:14.233667Z","submitted_at":"2025-05-31T04:31:02Z","title":"MagiCodec: Simple Masked Gaussian-Injected Codec for High-Fidelity Reconstruction and Generation","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T12:12:29.406719Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2506.00385"},"observation_digest":"sha256:953fafcb96845600ad7275b66e37083e93cdd3a86531a9da8741bd1313dff026","observation_id":"eeacbfb5-38ea-4628-8ef8-d397f3701c09","resolution":{"observed_at":"2026-08-07T12:12:29.406719Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-07T04:08:06.083636Z","title":"V ocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.11514","last_updated":"2025-06-13T07:15:41Z","snapshot_observed_at":"2026-08-18T17:51:38.775368Z","submitted_at":"2025-06-13T07:15:41Z","title":"Efficient Speech Enhancement via Embeddings from Pre-trained Generative Audioencoders","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T04:08:06.083636Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2506.11514"},"observation_digest":"sha256:31eb45c1a2d0ca8ef2ecbf9c80d48538f7f60b2445c143bdfb053d74f6b40622","observation_id":"b55dffe9-05b2-41cb-bde1-7f3dcca9ea0b","resolution":{"observed_at":"2026-08-07T04:08:06.083636Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-07T00:30:35.396683Z","title":"V ocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.13709","last_updated":"2025-06-16T17:19:04Z","snapshot_observed_at":"2026-08-21T11:44:22.271869Z","submitted_at":"2025-06-16T17:19:04Z","title":"SpeechRefiner: Towards Perceptual Quality Refinement for Front-End Algorithms","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T00:30:35.396683Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2506.13709"},"observation_digest":"sha256:013307a884186ba8dbe0aa847cb8068b5c94c1c6491eea9bac9fe11e4428ef99","observation_id":"513f7c56-5181-4066-ab7d-01fffc8e987d","resolution":{"observed_at":"2026-08-07T00:30:35.396683Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-06T21:51:35.862909Z","title":"V ocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.23325","last_updated":"2025-07-09T17:40:35Z","snapshot_observed_at":"2026-08-17T17:41:09.789585Z","submitted_at":"2025-06-29T16:51:50Z","title":"XY-Tokenizer: Mitigating the Semantic-Acoustic Conflict in Low-Bitrate Speech Codecs","version":2},"reference_index":2001,"source":"pdf_text","source_observed_at":"2026-08-06T21:51:35.862909Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2506.23325"},"observation_digest":"sha256:63fe8615a32e39c21e90216848f388d70de092122801bb607b29c2f952dceebf","observation_id":"16507183-20b4-4104-8119-91ffd070184e","resolution":{"observed_at":"2026-08-06T21:51:35.862909Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-06T20:06:18.832332Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.03887","last_updated":"2025-07-05T03:59:17Z","snapshot_observed_at":"2026-08-21T16:34:08.673603Z","submitted_at":"2025-07-05T03:59:17Z","title":"Traceable TTS: Toward Watermark-Free TTS with Strong Traceability","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T20:06:18.832332Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2507.03887"},"observation_digest":"sha256:d62c9f31341ad3796685eb421809ffc15133723cc1c41ef36cb36e7770663d8b","observation_id":"18874d1a-a21c-4683-8636-141d30755463","resolution":{"observed_at":"2026-08-06T20:06:18.832332Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-06T16:41:24.295671Z","title":"V ocos: Closing the gap between time-domain and fourier- based neural vocoders for high-quality audio synthesis,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.12825","last_updated":"2025-07-17T06:32:22Z","snapshot_observed_at":"2026-08-16T23:23:39.503713Z","submitted_at":"2025-07-17T06:32:22Z","title":"Autoregressive Speech Enhancement via Acoustic Tokens","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T16:41:24.295671Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2507.12825"},"observation_digest":"sha256:b22e238645bd1f7de3152ffab92c048c6087cb9f429c4cd5c06f1315760754e0","observation_id":"5b2c4a59-bc6d-40f7-8ffb-889dc66184bc","resolution":{"observed_at":"2026-08-06T16:41:24.295671Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-06T15:48:22.435245Z","title":"V ocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.14988","last_updated":"2025-07-20T14:48:48Z","snapshot_observed_at":"2026-08-11T00:57:14.977496Z","submitted_at":"2025-07-20T14:48:48Z","title":"DMOSpeech 2: Reinforcement Learning for Duration Prediction in Metric-Optimized Speech Synthesis","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:22.435245Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2507.14988"},"observation_digest":"sha256:7804347e938f116f010015cdff58018be78b306103dfc4cfab61cc8acdd54853","observation_id":"676782b7-0ac9-4013-aa6c-2f897b0f37d4","resolution":{"observed_at":"2026-08-06T15:48:22.435245Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-15T18:11:55.375110Z","title":"V ocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.18897","last_updated":"2025-07-25T02:44:30Z","snapshot_observed_at":"2026-08-20T13:35:55.682692Z","submitted_at":"2025-07-25T02:44:30Z","title":"HH-Codec: High Compression High-fidelity Discrete Neural Codec for Spoken Language Modeling","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-15T18:11:55.375110Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2507.18897"},"observation_digest":"sha256:682a4af98c7fa55036381242d44f22b1cf752092b1b3ffd7ce2ef629f3a54b95","observation_id":"9eb0fb14-fa1f-4604-a4f8-8687aee6342d","resolution":{"observed_at":"2026-08-15T18:11:55.375110Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":"2306.00814","doi":"10.48550/arxiv.2306.00814","metadata_source":"pith","pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis","venue":"cs.SD","work_id":"dfb28242-44a0-4f86-bb57-65dd1aae8947","year":2023},"citing_paper":{"arxiv_id":"2509.24708","last_updated":"2026-04-05T03:27:11Z","snapshot_observed_at":"2026-08-14T06:32:53.847982Z","submitted_at":"2025-09-29T12:34:58Z","title":"SenSE: Semantic-Aware High-Fidelity Universal Speech Enhancement","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-18T12:35:19.023045Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2509.24708"},"observation_digest":"sha256:93f69a6d56fef309c95a10c537b1514260bb6c5ef812d9adc4aca02c474b689b","observation_id":"68a7a26e-2885-4c50-9e47-7653efdfdfa9","resolution":{"observed_at":"2026-05-18T12:36:22.339938Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":"2306.00814","doi":"10.48550/arxiv.2306.00814","metadata_source":"pith","pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis","venue":"cs.SD","work_id":"dfb28242-44a0-4f86-bb57-65dd1aae8947","year":2023},"citing_paper":{"arxiv_id":"2512.01537","last_updated":"2026-05-18T17:15:31Z","snapshot_observed_at":"2026-08-15T21:40:35.377632Z","submitted_at":"2025-12-01T11:06:38Z","title":"Two-Dimensional Quantization for Geometry-Aware Audio Coding","version":3},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-05-21T18:16:51.486807Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2512.01537"},"observation_digest":"sha256:d46866117d0b94783a5890f3678f35e10bd4f173ea40097ff7be0be13df2e682","observation_id":"62936d30-161f-4588-b923-eb3c1dbe8c65","resolution":{"observed_at":"2026-05-21T18:20:29.314335Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":"2306.00814","doi":"10.48550/arxiv.2306.00814","metadata_source":"pith","pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis","venue":"cs.SD","work_id":"dfb28242-44a0-4f86-bb57-65dd1aae8947","year":2023},"citing_paper":{"arxiv_id":"2604.01929","last_updated":"2026-04-29T10:54:09Z","snapshot_observed_at":"2026-07-06T22:51:40.181565Z","submitted_at":"2026-04-02T11:49:00Z","title":"Woosh: A Sound Effects Foundation Model","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-13T20:51:08.144573Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2604.01929"},"observation_digest":"sha256:2cbbc2effe14ac7854886df8cb73061e84175a6389d98360c83691cb84558f5d","observation_id":"178320c8-2265-497d-8752-ec691fc40ac4","resolution":{"observed_at":"2026-05-13T20:53:15.813464Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":"2306.00814","doi":"10.48550/arxiv.2306.00814","metadata_source":"pith","pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis","venue":"cs.SD","work_id":"dfb28242-44a0-4f86-bb57-65dd1aae8947","year":2023},"citing_paper":{"arxiv_id":"2604.14606","last_updated":"2026-07-20T08:55:18Z","snapshot_observed_at":"2026-08-12T23:58:03.604754Z","submitted_at":"2026-04-16T04:25:03Z","title":"UniPASE: A Generative Model for Universal Speech Enhancement with High Fidelity and Low Hallucinations","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-10T10:18:16.972414Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2604.14606"},"observation_digest":"sha256:08a259c3a28611a69a6c516841af5a82b78416b13e4c532317f51cae0cf0a13a","observation_id":"a0c32f32-2706-4912-86a0-4f01aa865198","resolution":{"observed_at":"2026-05-10T10:19:19.976408Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-02T16:16:37.406485Z","title":"V ocos: Closing the gap between time-domain and fourier- based neural vocoders for high-quality audio synthesis,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2604.14606","last_updated":"2026-07-20T08:55:18Z","snapshot_observed_at":"2026-08-12T23:58:03.604754Z","submitted_at":"2026-04-16T04:25:03Z","title":"UniPASE: A Generative Model for Universal Speech Enhancement with High Fidelity and Low Hallucinations","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-02T16:16:37.406485Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2604.14606"},"observation_digest":"sha256:96fafdbab1cfed787fb9dad7a4c410c660ea38301650dc91d08529b44050b26b","observation_id":"9a992810-63b1-45e6-81b4-95ddc09f16ff","resolution":{"observed_at":"2026-08-02T16:16:37.406485Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":"2306.00814","doi":"10.48550/arxiv.2306.00814","metadata_source":"pith","pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis","venue":"cs.SD","work_id":"dfb28242-44a0-4f86-bb57-65dd1aae8947","year":2023},"citing_paper":{"arxiv_id":"2606.03455","last_updated":"2026-06-02T10:33:20Z","snapshot_observed_at":"2026-08-05T18:48:21.902772Z","submitted_at":"2026-06-02T10:33:20Z","title":"WavTTS: Towards High-Quality Zero-Shot TTS via Direct Raw Waveform Modeling","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-06-28T08:18:42.002083Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2606.03455"},"observation_digest":"sha256:723347f3c57fbb38be6b4da8af0fd94021a7ad66ea07ac68f678fa02532cdab5","observation_id":"311e960d-e47a-4882-9ac5-9415515192ac","resolution":{"observed_at":"2026-07-02T05:16:39.843462Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":"2306.00814","doi":"10.48550/arxiv.2306.00814","metadata_source":"pith","pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis","venue":"cs.SD","work_id":"dfb28242-44a0-4f86-bb57-65dd1aae8947","year":2023},"citing_paper":{"arxiv_id":"2606.09048","last_updated":"2026-06-08T05:36:42Z","snapshot_observed_at":"2026-07-06T23:48:26.076260Z","submitted_at":"2026-06-08T05:36:42Z","title":"BareWave: Waveform-Native Flow-Matching Text-to-Speech","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-27T15:14:03.681833Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2606.09048"},"observation_digest":"sha256:d1ea967fd0bf47ed5e5f81cbef21133dd669b716063e5782ab18f302f49fbc85","observation_id":"648201a4-36d9-4773-a494-e62e2afccb2f","resolution":{"observed_at":"2026-07-03T03:37:35.201518Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":"2306.00814","doi":"10.48550/arxiv.2306.00814","metadata_source":"pith","pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis","venue":"cs.SD","work_id":"dfb28242-44a0-4f86-bb57-65dd1aae8947","year":2023},"citing_paper":{"arxiv_id":"2606.11219","last_updated":"2026-05-11T20:27:40Z","snapshot_observed_at":"2026-08-05T20:46:41.277498Z","submitted_at":"2026-05-11T20:27:40Z","title":"Afrispeech Semantics: Evaluating Audio Semantic Reasoning in Spoken Language Models Across Domains and Accents","version":1},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-06-30T22:11:44.891731Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2606.11219"},"observation_digest":"sha256:eec52bcd512f8f012444fe9361b50657a8c8061d78bcdc9d4a41e5dfefefe3c6","observation_id":"99123b19-0579-4663-b5b5-e39575b762bb","resolution":{"observed_at":"2026-06-30T22:15:05.635697Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":"2306.00814","doi":"10.48550/arxiv.2306.00814","metadata_source":"pith","pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis","venue":"cs.SD","work_id":"dfb28242-44a0-4f86-bb57-65dd1aae8947","year":2023},"citing_paper":{"arxiv_id":"2606.11611","last_updated":"2026-06-10T03:23:21Z","snapshot_observed_at":"2026-08-14T10:11:40.647550Z","submitted_at":"2026-06-10T03:23:21Z","title":"SARA: A Dual-Stream VAE for High-Fidelity Speech Generation via Integrating Semantic and Acoustic Representations","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-27T08:38:51.371500Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2606.11611"},"observation_digest":"sha256:b61ad408edb90658cc64fa89c9fd98ef266473fcb4a2d5d7e03512f9e0497825","observation_id":"6c09b5b2-5213-461a-bda2-ccf89aa39fb3","resolution":{"observed_at":"2026-07-03T12:58:08.358364Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":"2306.00814","doi":"10.48550/arxiv.2306.00814","metadata_source":"pith","pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis","venue":"cs.SD","work_id":"dfb28242-44a0-4f86-bb57-65dd1aae8947","year":2023},"citing_paper":{"arxiv_id":"2606.17806","last_updated":"2026-06-16T11:28:29Z","snapshot_observed_at":"2026-08-09T03:28:22.287482Z","submitted_at":"2026-06-16T11:28:29Z","title":"PhASE-Flow: Phonetic-Conditioned Acoustic Flow Matching in SSL Representation Domain for Speech Enhancement","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-26T23:05:08.894347Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2606.17806"},"observation_digest":"sha256:07685253444a663fa3b299fe42cb1bd6d4d101a338429341981b54cee26a47a8","observation_id":"8dec23c5-03d7-4633-a10e-3d92de0d2bad","resolution":{"observed_at":"2026-07-03T23:09:00.912005Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":"2306.00814","doi":"10.48550/arxiv.2306.00814","metadata_source":"pith","pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis","venue":"cs.SD","work_id":"dfb28242-44a0-4f86-bb57-65dd1aae8947","year":2023},"citing_paper":{"arxiv_id":"2606.21372","last_updated":"2026-06-19T12:24:49Z","snapshot_observed_at":"2026-08-20T19:17:49.674851Z","submitted_at":"2026-06-19T12:24:49Z","title":"NAC: Neural Action Codec for Vision-Language-Action Models","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-06-26T13:59:53.484305Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2606.21372"},"observation_digest":"sha256:780b93fe54fe98f5a9e0fa61cfcb2493cc0c3ee404486e05dee2de124cc56dc2","observation_id":"432d4ae4-8932-40ba-b173-96e958d3901b","resolution":{"observed_at":"2026-07-04T06:59:37.890291Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":"2306.00814","doi":"10.48550/arxiv.2306.00814","metadata_source":"pith","pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis","venue":"cs.SD","work_id":"dfb28242-44a0-4f86-bb57-65dd1aae8947","year":2023},"citing_paper":{"arxiv_id":"2606.23176","last_updated":"2026-06-22T11:14:22Z","snapshot_observed_at":"2026-08-19T04:10:47.042253Z","submitted_at":"2026-06-22T11:14:22Z","title":"Synthesizing the Lombard Effect: Multi-Level Control of Speech Clarity and Vocal Effort in TTS","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-06-26T06:49:45.405554Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2606.23176"},"observation_digest":"sha256:af14b6e7eaa1a6e6bd5d4716756c1bb65cb3490189fdda7636d7f2529d54eff1","observation_id":"662649d3-2a3e-464e-aed4-30e1c9418328","resolution":{"observed_at":"2026-07-04T12:29:51.743004Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":"2306.00814","doi":"10.48550/arxiv.2306.00814","metadata_source":"pith","pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis","venue":"cs.SD","work_id":"dfb28242-44a0-4f86-bb57-65dd1aae8947","year":2023},"citing_paper":{"arxiv_id":"2606.23190","last_updated":"2026-07-08T11:35:21Z","snapshot_observed_at":"2026-08-12T18:07:30.628246Z","submitted_at":"2026-06-22T11:37:45Z","title":"FlowTTS-GRPO: Online Reinforcement Learning with Multi-Objective Reward Optimization for Flow-Matching Based Text-to-Speech","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-26T07:02:36.499424Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2606.23190"},"observation_digest":"sha256:aa2a9eb6d91f335e13e2456bcf30f6f5994717c59e2930c61b1b69e1149939b4","observation_id":"cd3727ec-abf2-4b56-94b8-836d6b00c7f6","resolution":{"observed_at":"2026-07-04T12:19:49.654721Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-07-12T12:44:20.831164Z","title":"V ocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.23190","last_updated":"2026-07-08T11:35:21Z","snapshot_observed_at":"2026-08-12T18:07:30.628246Z","submitted_at":"2026-06-22T11:37:45Z","title":"FlowTTS-GRPO: Online Reinforcement Learning with Multi-Objective Reward Optimization for Flow-Matching Based Text-to-Speech","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-07-12T12:44:20.831164Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2606.23190"},"observation_digest":"sha256:95415f96558e4aeae66ce22a320ad6cb4db820fe36e01ca635a2aaa0c9a87f88","observation_id":"4c3a6dcd-7094-4446-ad5e-af2b5c4292ab","resolution":{"observed_at":"2026-07-12T12:44:20.831164Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":"2306.00814","doi":"10.48550/arxiv.2306.00814","metadata_source":"pith","pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis","venue":"cs.SD","work_id":"dfb28242-44a0-4f86-bb57-65dd1aae8947","year":2023},"citing_paper":{"arxiv_id":"2606.31247","last_updated":"2026-06-30T07:24:10Z","snapshot_observed_at":"2026-08-13T09:35:09.713263Z","submitted_at":"2026-06-30T07:24:10Z","title":"FlexiSLM: A Dynamic and Controllable Frame Rate Spoken Language Model","version":1},"reference_index":116,"source":"arxiv_source","source_observed_at":"2026-07-01T03:50:26.873406Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2606.31247"},"observation_digest":"sha256:5107f08c059bd290dd3f608ada77eb53a020f15040d5f623e2d568cccae40c99","observation_id":"3d2018f4-9fdd-48aa-9fc2-f2f9f6ddbafc","resolution":{"observed_at":"2026-07-01T11:55:42.189973Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":"2306.00814","doi":"10.48550/arxiv.2306.00814","metadata_source":"pith","pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis","venue":"cs.SD","work_id":"dfb28242-44a0-4f86-bb57-65dd1aae8947","year":2023},"citing_paper":{"arxiv_id":"2607.05196","last_updated":"2026-07-07T15:36:48Z","snapshot_observed_at":"2026-08-19T04:43:08.967042Z","submitted_at":"2026-07-06T15:11:57Z","title":"Unified Audio Intelligence Without Regressing on Text Intelligence","version":1},"reference_index":167,"source":"arxiv_source","source_observed_at":"2026-07-07T23:59:38.702609Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2607.05196"},"observation_digest":"sha256:0dae801fcdba78f471db834b04dd2b129d941502b0bfc8efa768a6b5070ee302","observation_id":"898e4653-2033-4961-a5d5-9e5126652721","resolution":{"observed_at":"2026-07-08T00:04:22.674104Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-07-11T07:46:49.059192Z","title":"arXiv preprint arXiv:2306.00814 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.05196","last_updated":"2026-07-07T15:36:48Z","snapshot_observed_at":"2026-08-19T04:43:08.967042Z","submitted_at":"2026-07-06T15:11:57Z","title":"Unified Audio Intelligence Without Regressing on Text Intelligence","version":2},"reference_index":167,"source":"arxiv_source","source_observed_at":"2026-07-11T07:46:49.059192Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2607.05196"},"observation_digest":"sha256:0cfa54cd57da6a725eefa25d5a2bec04fec507f5cc69ebacf322377f5a435772","observation_id":"824fa72a-15bd-48fd-ae54-090373c1d9e3","resolution":{"observed_at":"2026-07-11T07:46:49.059192Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-02T06:31:32.258567Z","title":"V ocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.12496","last_updated":"2026-07-15T05:26:04Z","snapshot_observed_at":"2026-08-20T09:02:05.038840Z","submitted_at":"2026-07-14T08:28:48Z","title":"ZipL-Dialog: Memory-Efficient Long-Form Spoken Dialog Synthesis via Latent Flow Matching","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-02T06:31:32.258567Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2607.12496"},"observation_digest":"sha256:5ffb0eac244489a95816bbf9f867ea201fbde30421c3b0fd5c9e27d8c6dc8bb5","observation_id":"cf042f65-49d9-4778-b6aa-65d59d4f3b99","resolution":{"observed_at":"2026-08-02T06:31:32.258567Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-01T18:46:14.038134Z","title":"Vocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis.arXiv preprint arXiv:2306.00814, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.18662","last_updated":"2026-07-19T11:20:25Z","snapshot_observed_at":"2026-08-20T16:02:45.480524Z","submitted_at":"2026-07-19T11:20:25Z","title":"Staged Depth-Pruning Distillation of a Flow-Matching Text-to-Speech Teacher: A Compact Hindi Speech Synthesizer","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-01T18:46:14.038134Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2607.18662"},"observation_digest":"sha256:67b7c1d63d9a286c3f63d55bb87784cefe94221fe15ff5c62f9e417f45c63d4b","observation_id":"4922feff-a1cd-4b7e-97b4-377f7cebfe07","resolution":{"observed_at":"2026-08-01T18:46:14.038134Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-16T00:37:04.982405Z","title":"V ocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis.arXiv preprint arXiv:2306.00814, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.11737","last_updated":"2026-08-12T07:23:54Z","snapshot_observed_at":"2026-08-18T22:02:39.463120Z","submitted_at":"2026-08-12T07:23:54Z","title":"Phoenix TTS: High-Fidelity Synthesis and Voice Conversion via Flow-Matching-Driven Speech Tokenization","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-16T00:37:04.982405Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2608.11737"},"observation_digest":"sha256:b05c3e2cc442977375292c269f00c47cbe4da64d1fddcfe70ad1dee5a3bca7d5","observation_id":"bef0a40f-7795-4b00-ac2a-a536a6ff8af7","resolution":{"observed_at":"2026-08-16T00:37:04.982405Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2306.00814/citation-record","integrity":"/paper/2306.00814/integrity","json":"/paper/2306.00814/citation-record.json","paper":"/paper/2306.00814"},"outbound":[],"paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","latest_version":3,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-18T17:50:53.134857Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"thesis":"As of 23 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 40 inbound Pith citation observations for arXiv:2306.00814."}