{"as_of":"2026-08-11T13:46:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:783c66be539cf039cd9b7ee7ebbfcde9a04ceb57b9f44673ffbb41b38c220283","coverage":[{"denominator":35,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":35,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T16:15:49.630165Z","state":"measured"},{"denominator":35,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":35,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2501.13372/citation-record","integrity":"/paper/2501.13372/integrity","json":"/paper/2501.13372/citation-record.json","paper":"/paper/2501.13372"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:50.100731Z","title":"Robust speech recognition via large-scale weak supervision,","venue":null,"work_id":"4d71e3ad-921f-4080-9cca-662300b456bd","year":2023},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.506118Z"},"links":{"citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:6f321c9ae8df6f3707e021c181dec55f39ab9fa93783ff100d8e85350f1cbfe0","observation_id":"a8b56282-4bae-46e3-bc66-06c80343c1bf","resolution":{"observed_at":"2026-08-10T16:15:50.105320Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:50.089254Z","title":"Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech,","venue":null,"work_id":"88d3192c-a874-48df-b987-c802c43b2f7a","year":2021},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.510580Z"},"links":{"citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:b2c787ab3913b9e04b78efa58a4b55a110a9295520e2532a9d996bfe9fb8aa97","observation_id":"ace06ba6-00a3-471c-8448-0f93efb8381f","resolution":{"observed_at":"2026-08-10T16:15:50.093576Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:50.079292Z","title":"YourTTS: Towards zero-shot multi-speaker TTS and zero-shot voice conversion for everyone,","venue":null,"work_id":"0d333858-da14-453f-bed7-5151b7fd41e2","year":2022},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.514236Z"},"links":{"citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:14e97ff0085fd84b9fc00a65eaab0cac9b4ace11395e62580049708281a1844e","observation_id":"556021c8-0bb1-4cf2-8e40-483d685a34c2","resolution":{"observed_at":"2026-08-10T16:15:50.082843Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:50.069107Z","title":"XTTS: A massively multilingual zero-shot text-to-speech model,","venue":null,"work_id":"acf210ce-694b-450d-86cd-d08e08566c06","year":2024},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.518454Z"},"links":{"citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:6d458f6091778d18521bfc3adce24c00be2a92665f2fb6ae94d8ba2e5a201eba","observation_id":"f11c3f42-3efe-4103-8466-bbfd6fe585c5","resolution":{"observed_at":"2026-08-10T16:15:50.072279Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.02111","last_updated":"2023-01-05T15:37:15Z","snapshot_observed_at":"2026-08-07T10:11:17.796562Z","submitted_at":"2023-01-05T15:37:15Z","title":"Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.02111","snapshot_observed_at":"2026-08-10T16:15:49.521891Z","title":"Neural codec language models are zero-shot text to speech synthesizers,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.521891Z"},"links":{"cited_paper":"/paper/2301.02111","citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:db2f2116b9efe3c8ec402f3c7927f17911ab981fea5067052c9a22bc62e578a4","observation_id":"76e75cc2-3f2f-4cc7-baf1-8af9735ac706","resolution":{"observed_at":"2026-08-10T16:15:49.521891Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:50.059370Z","title":"Metric- GAN+: An improved version of MetricGAN for speech enhancement,","venue":null,"work_id":"8afb1b93-c53c-4761-9682-4a38c83b44c9","year":2021},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.526026Z"},"links":{"citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:30230928c7450436649cc7327404246c7f63dd8144bcc09af072a15ca0812bdb","observation_id":"0c1a3863-287c-4e2b-96b5-671ca32e8c17","resolution":{"observed_at":"2026-08-10T16:15:50.062785Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:50.049287Z","title":"MANNER: Multi-view attention netwfork for noise erasure,","venue":null,"work_id":"ab529041-7f60-455b-a6b8-b9608dfa3790","year":2022},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.529713Z"},"links":{"citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:fd7702faaa6088ef05d80aadf7126c01e314a269a5f130eab100777db874eda8","observation_id":"180b4d37-40be-4afe-9aa4-d2572b0d894e","resolution":{"observed_at":"2026-08-10T16:15:50.052797Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:50.036960Z","title":"Text is all you need: Personalizing ASR models using controllable speech synthesis,","venue":null,"work_id":"636a047d-d991-4d16-a2fb-ff8b41caf23c","year":2023},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.532947Z"},"links":{"citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:5599dc658abee90655890181d23e57592fec1624bc549f4acda17286e8e56b19","observation_id":"3920a48e-359d-4083-a7e3-8ae60ca84acf","resolution":{"observed_at":"2026-08-10T16:15:50.041005Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:50.025313Z","title":"The potential of neural speech synthesis-based data augmentation for personalized speech en- hancement,","venue":null,"work_id":"8812d832-6909-4c00-80f5-c0677852cd35","year":2023},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.536053Z"},"links":{"citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:e0bba8b02b2c1bfd638b77d79cd0ea8f7235e4dcb40bd3e6abfcf51d33589a94","observation_id":"e5f5fa13-285e-4f18-9eb8-3c28b9f06753","resolution":{"observed_at":"2026-08-10T16:15:50.029364Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:50.007804Z","title":"A survey on image data augmen- tation for deep learning,","venue":null,"work_id":"57eb9b74-ae78-4ab1-bf2f-7c37c76f5b5d","year":2019},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.539283Z"},"links":{"citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:f4a0a257799721f8a6d6bbfdc659ce9cd8c558c26b60fa27b0d74f6f6c746d81","observation_id":"171e1ed6-028a-48ff-ba2d-94b149dcaaa7","resolution":{"observed_at":"2026-08-10T16:15:50.017399Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:49.994965Z","title":"Effec- tive data augmentation with diffusion models,","venue":null,"work_id":"f54cf02e-b999-4585-a9c4-1ded0d63b782","year":2024},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.542674Z"},"links":{"citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:e59c246701872ff448dd7ee0a664efe3c3795b1c37e0078ea02981773b54794b","observation_id":"bdb7f0cb-e38c-4ef1-8560-fc0930e0f4ed","resolution":{"observed_at":"2026-08-10T16:15:49.999133Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:49.982922Z","title":"SpecAug- ment: A simple data augmentation method for automatic speech recog- nition,","venue":null,"work_id":"29e2af07-6da1-4a55-a15d-6179e35da73d","year":2019},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.546142Z"},"links":{"citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:0a426cf3663e6ee8a1f904895beb892452a8314a9cf50f3bc6e138b622f61923","observation_id":"1e7ab5a0-53d2-4c7c-88e1-8644a2f3c81a","resolution":{"observed_at":"2026-08-10T16:15:49.987157Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:49.971656Z","title":"Latent filling: Latent space data augmentation for zero-shot speech synthesis,","venue":null,"work_id":"46345abf-b8dc-4edc-bc62-af338d8472f0","year":2024},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.549214Z"},"links":{"citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:3326cef4ac999e2b3bad4e8cec03919899bb24f8b99b9b5273e7b5f02181e44d","observation_id":"01399470-8c03-4018-a842-ada66b1a92fb","resolution":{"observed_at":"2026-08-10T16:15:49.975500Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:49.961886Z","title":"Improving few-shot learning for talking face system with TTS data augmentation,","venue":null,"work_id":"e2c401e7-dcb0-4e3b-90d1-88f3d0eca66d","year":2023},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.552369Z"},"links":{"citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:7932f3bc6cf267b3a31a281bc3c83711318386278a416361f85c2678c0edd89f","observation_id":"51fd7b0b-799e-44db-bfac-039c883e2769","resolution":{"observed_at":"2026-08-10T16:15:49.965190Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.48550/arxiv.2407.18879","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:49.756267Z","title":"Utilizing TTS synthesized data for efficient development of keyword spotting model,","venue":null,"work_id":"19f449a5-ce86-4d41-bcdf-15eab575c230","year":2024},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.555491Z"},"links":{"citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:91d4f642f551523e2a78f2568c187e6dd054452580aadfac021cd050aad63e95","observation_id":"9d0549c9-50ff-473d-9843-c3a2659cc45d","resolution":{"observed_at":"2026-08-10T16:15:49.761403Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:49.952311Z","title":"Zero shot text to speech augmentation for automatic speech recognition on low-resource accented speech corpora,","venue":null,"work_id":"e8116bd3-8427-4b08-9d38-0020844aeb86","year":2023},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.558899Z"},"links":{"citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:ec069640c844e846ff902d5ccb1e63e3216e4e7a6e27d4d950e1d2ed37caf352","observation_id":"2d5ad68b-3911-4680-a076-0fba34876169","resolution":{"observed_at":"2026-08-10T16:15:49.955486Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:49.941555Z","title":"LibriTTS: A corpus derived from librispeech for text-to-speech,","venue":null,"work_id":"75d8a358-4918-413a-a0f5-b28267085640","year":2019},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.562534Z"},"links":{"citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:42501c6e23ae7d2f92d3e5d6ba5f7cbbe07b83227d155c7c1fd929974d066f5b","observation_id":"212694fd-6d07-43c8-9e2e-a203a706cf36","resolution":{"observed_at":"2026-08-10T16:15:49.945620Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1510.08484","last_updated":"2015-10-28T20:59:04Z","snapshot_observed_at":"2026-07-06T04:34:36.474437Z","submitted_at":"2015-10-28T20:59:04Z","title":"MUSAN: A Music, Speech, and Noise Corpus","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1510.08484","snapshot_observed_at":"2026-08-10T16:15:49.566104Z","title":"MUSAN: A music, speech, and noise corpus,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.566104Z"},"links":{"cited_paper":"/paper/1510.08484","citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:8c302868c882a911024d0367e81a49f95dc8ec45d637616d1b826dad9b8500f9","observation_id":"9d3977ab-c3a0-4707-8268-821acafd00f4","resolution":{"observed_at":"2026-08-10T16:15:49.566104Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.04624","last_updated":"2021-06-08T18:22:56Z","snapshot_observed_at":"2026-08-04T07:20:14.379561Z","submitted_at":"2021-06-08T18:22:56Z","title":"SpeechBrain: A General-Purpose Speech Toolkit","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.04624","snapshot_observed_at":"2026-08-10T16:15:49.569709Z","title":"SpeechBrain: A general-purpose speech toolkit,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.569709Z"},"links":{"cited_paper":"/paper/2106.04624","citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:b08f382f9e590eab8ff3489c98ddf1a29956535abface8124238361526f7e053","observation_id":"8b78eaca-73ef-45c9-aadc-bd0af01f1c44","resolution":{"observed_at":"2026-08-10T16:15:49.569709Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:49.930650Z","title":"Deep MOS predictor for synthetic speech using cluster-based modeling,","venue":null,"work_id":"d9351c4e-38b9-4575-a980-035c7b202c2d","year":2020},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.573551Z"},"links":{"citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:daaf6e670e01c8c0e421cf4c9e12de011ab2967e72a06387ed0ef7b547b06483","observation_id":"fbb3bceb-93f3-4cd5-ab94-d3f9226851d5","resolution":{"observed_at":"2026-08-10T16:15:49.934442Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:49.918973Z","title":"NISQA: A deep CNN- self-attention model for multidimensional speech quality prediction with crowdsourced datasets,","venue":null,"work_id":"ebc15359-c99d-437d-8ae4-4ab463779539","year":2021},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.577135Z"},"links":{"citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:7c7cec7e7fcb780a76e492e6448290af8adc8c2621b8a4aa14057ffa2e2056a4","observation_id":"625a1673-ed59-4411-8211-9ad2eed1aa11","resolution":{"observed_at":"2026-08-10T16:15:49.923282Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:49.907490Z","title":"The singing voice conversion challenge 2023,","venue":null,"work_id":"d1c649a2-8293-4020-8936-7dcdac98b40c","year":2023},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.581878Z"},"links":{"citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:79aeda46092f040b5999d80bfb64d1cae6aa6e39d1109434e183622de3b6a018","observation_id":"c795598e-c890-4187-8a0e-f3216730d570","resolution":{"observed_at":"2026-08-10T16:15:49.911509Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:49.894984Z","title":"NaturalSpeech 3: Zero- shot speech synthesis with factorized codec and diffusion models,","venue":null,"work_id":"c9e7d01f-5e47-4323-8695-0a5d4f01a270","year":2024},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.585624Z"},"links":{"citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:77d5787939a752ae442c13fc0784fcf70ae16307b1a03a01edf35fd24b0e2307","observation_id":"c639dc55-a213-4604-ac35-39f4e088b1c0","resolution":{"observed_at":"2026-08-10T16:15:49.898964Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.09305","last_updated":"2024-09-14T05:03:18Z","snapshot_observed_at":"2026-07-06T19:15:17.200627Z","submitted_at":"2024-09-14T05:03:18Z","title":"The T05 System for The VoiceMOS Challenge 2024: Transfer Learning from Deep Image Classifier to Naturalness MOS Prediction of High-Quality Synthetic Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.09305","snapshot_observed_at":"2026-08-10T16:15:49.589322Z","title":"The T05 system for the V oiceMOS challenge 2024: Transfer learning from deep image classifier to naturalness MOS prediction of high-quality synthetic speech,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.589322Z"},"links":{"cited_paper":"/paper/2409.09305","citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:def0dde9faa9e0649d8581284f740d35a8d6aa59bc52e85444b33305de6a409a","observation_id":"890cdb57-1aff-4c5e-bdb1-09b130093fd9","resolution":{"observed_at":"2026-08-10T16:15:49.589322Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.07001","last_updated":"2024-09-11T04:26:38Z","snapshot_observed_at":"2026-07-06T19:13:33.255300Z","submitted_at":"2024-09-11T04:26:38Z","title":"The VoiceMOS Challenge 2024: Beyond Speech Quality Prediction","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.07001","snapshot_observed_at":"2026-08-10T16:15:49.593554Z","title":"The V oiceMOS challenge 2024: Beyond speech quality prediction,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.593554Z"},"links":{"cited_paper":"/paper/2409.07001","citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:e703553df546c3b24aa459cde0b183548e2aae229c63035bfb41646702d03db1","observation_id":"08c96b53-870f-4f32-9512-dc95a7f84e63","resolution":{"observed_at":"2026-08-10T16:15:49.593554Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:49.883274Z","title":"Performance measurement in blind audio source separation,","venue":null,"work_id":"25273ac7-3cc0-4115-a6c5-04832d94630d","year":2006},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.598175Z"},"links":{"citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:1215573631e54dc3b3558ad07b1f4983706fce448295d796fd89d3d1e08a719e","observation_id":"db4be3dd-d394-487f-9bed-8960f3d47368","resolution":{"observed_at":"2026-08-10T16:15:49.887332Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:49.869637Z","title":"A short- time objective intelligibility measure for time-frequency weighted noisy speech,","venue":null,"work_id":"84b12846-6f65-438b-bbee-521dd24a77fc","year":2010},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.601809Z"},"links":{"citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:2a733050a9f023450f0171c03a83ea0a0691ae29842d742e260a1957709e3df1","observation_id":"c6554e69-ab31-4d57-8fdc-c7bd0c94e3da","resolution":{"observed_at":"2026-08-10T16:15:49.874560Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:49.857950Z","title":"Perceptual evaluation of speech quality (PESQ)-a new method for speech quality assessment of telephone networks and codecs,","venue":null,"work_id":"00c432ba-aff7-4fd4-83bb-8a0395c21dd1","year":2001},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.605662Z"},"links":{"citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:870cce6ed59aa7c83f1d9633d0a70e6993eb0c972cd0994154815e442b13298c","observation_id":"1dbafe63-af65-4ed3-8756-baff11fb2159","resolution":{"observed_at":"2026-08-10T16:15:49.861940Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:49.846417Z","title":"SpeechT5: Unified- modal encoder-decoder pre-training for spoken language processing,","venue":null,"work_id":"4eb13a19-8dde-465d-b46f-340c9e982ed9","year":2022},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.609384Z"},"links":{"citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:11f399c0658165f3b3760e2696a4355fe0eb697c8f223d7ef8531642692daff7","observation_id":"c6a2707d-6491-4b1b-a958-6aaa0f69ce91","resolution":{"observed_at":"2026-08-10T16:15:49.849985Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07243","last_updated":"2023-05-23T21:41:54Z","snapshot_observed_at":"2026-08-10T19:02:23.351550Z","submitted_at":"2023-05-12T04:19:49Z","title":"Better speech synthesis through scaling","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.07243","snapshot_observed_at":"2026-08-10T16:15:49.613067Z","title":"Better speech synthesis through scaling,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.613067Z"},"links":{"cited_paper":"/paper/2305.07243","citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:7bc763327d303b10b292d56c79ac3e28da16f4b0c9881629ada40688cb88f5f3","observation_id":"e5b17c5a-2f26-4484-8557-15f96cbfbb5f","resolution":{"observed_at":"2026-08-10T16:15:49.613067Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:49.617374Z","title":"Conv-TasNet: Surpassing ideal time–frequency magnitude masking for speech separation,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.617374Z"},"links":{"citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:41f6772d1ea3359fc53ce145cdbcd7f1f757d58182d5e730ed4674a8aae7d030","observation_id":"3d8ff40b-ea82-4085-9bc9-32a6358b1e7c","resolution":{"observed_at":"2026-08-10T16:15:49.617374Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:49.827340Z","title":"Efficient personalized speech enhancement through self-supervised learning,","venue":null,"work_id":"0919cdba-70e1-4fb2-8b0a-bc9af3d7e3b7","year":2022},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.620473Z"},"links":{"citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:c5f2ca6c9a6510efdb0e81d54f819831986ced654c5f99423cf87754fd55fe37","observation_id":"98817ee4-eb23-4804-ae9e-80c760431438","resolution":{"observed_at":"2026-08-10T16:15:49.831804Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:49.816678Z","title":"Librispeech: An ASR corpus based on public domain audio books,","venue":null,"work_id":"93e4f056-3ee7-464d-944a-883407603b29","year":2015},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.623970Z"},"links":{"citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:0b7954d50a6d9ca13b6d5103acce4fd30591a78e7125247d85bb131dfbe9c904","observation_id":"45348eb1-c64e-4037-8f54-614b632690f6","resolution":{"observed_at":"2026-08-10T16:15:49.820175Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:49.627093Z","title":"FSD50K: An open dataset of human-labeled sound events,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.627093Z"},"links":{"citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:27745b9104cc09e291f09f80223eb6fd51d6e6ac7bd91aba2ab10109a93563cd","observation_id":"c76526c2-9544-4905-ae19-7cdeef27dc3c","resolution":{"observed_at":"2026-08-10T16:15:49.627093Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1412.6980","last_updated":"2017-01-30T01:27:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2014-12-22T13:54:29Z","title":"Adam: A Method for Stochastic Optimization","version":9},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1412.6980","snapshot_observed_at":"2026-08-10T16:15:49.630165Z","title":"Adam: A method for stochastic optimization,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-10T16:15:49.630165Z"},"links":{"cited_paper":"/paper/1412.6980","citing_paper":"/paper/2501.13372"},"observation_digest":"sha256:1eaca607e12b9dcb31db67a4e7996e2997c4f78559c10101fec6ea424049b209","observation_id":"491f2fd1-750f-431c-9def-71cab089c4d5","resolution":{"observed_at":"2026-08-10T16:15:49.630165Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2501.13372","last_updated":"2025-01-23T04:27:37Z","latest_version":1,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-10T19:02:32.813870Z","submitted_at":"2025-01-23T04:27:37Z","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement"},"reference_resolution":{"displayed":35,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":9,"verified_exact":1,"verified_fuzzy":25},"total_outbound_references":35},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 35 of 35 outbound references and 0 inbound Pith citation observations for arXiv:2501.13372."}