{"as_of":"2026-08-10T11:22:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:34f55dbe713a8decc3b7c9b5e861f86f1983bd3db9ac238acbfb81031785aaa6","coverage":[{"denominator":69,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":69,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T20:40:16.437239Z","state":"measured"},{"denominator":70,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":70,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T20:40:10.960719Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-06T20:40:17.636566Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"cited_work":{"arxiv_id":"2507.02176","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.02176","snapshot_observed_at":"2026-08-06T20:40:17.636566Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","venue":"cs.SD","work_id":"722c863c-9881-4c3d-b685-3930caa74464","year":2025},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:10.960719Z"},"links":{"cited_paper":"/paper/2507.02176","citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:f58ac9a1de358b72ff8265e98a10608d629bbcbeb7260384640dd1d46ebe7e29","observation_id":"517abf04-5934-49ef-95b2-e23ab2515c83","resolution":{"observed_at":"2026-08-06T20:40:17.833571Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.02176/citation-record","integrity":"/paper/2507.02176/integrity","json":"/paper/2507.02176/citation-record.json","paper":"/paper/2507.02176"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:29.433403Z","title":"Virtual chara cters are expected to possess unique identities that remain consi stent across time","venue":null,"work_id":"cdec66a7-04a6-4952-9a5a-cc6ce1d70b80","year":null},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:10.892442Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:99dde1bc98b4629535a8265cb7fda039067f22f694d89a579f022ae9f5c931fc","observation_id":"fbd3ef56-0b7a-4d69-8241-d50fb04b296f","resolution":{"observed_at":"2026-08-06T20:40:29.587245Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"cited_work":{"arxiv_id":"2507.02176","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.02176","snapshot_observed_at":"2026-08-06T20:40:17.636566Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","venue":"cs.SD","work_id":"722c863c-9881-4c3d-b685-3930caa74464","year":2025},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:10.960719Z"},"links":{"cited_paper":"/paper/2507.02176","citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:f58ac9a1de358b72ff8265e98a10608d629bbcbeb7260384640dd1d46ebe7e29","observation_id":"517abf04-5934-49ef-95b2-e23ab2515c83","resolution":{"observed_at":"2026-08-06T20:40:17.833571Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:28.563475Z","title":"We explore which mark- ers are represented in some widely used ASV embeddings, and measure the effect of confounding factors","venue":null,"work_id":"2f84d85b-f750-41d0-8a43-2072a98252dd","year":2023},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:11.033440Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:62a64e0ea3fd1c98ba0c04e61271e6498ce064e22e8a9a1b985b04b397003ed7","observation_id":"f1f5725e-124c-47c6-963b-359f243ac434","resolution":{"observed_at":"2026-08-06T20:40:28.996348Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:28.203914Z","title":"This reﬂects the lower bound of our metric, where we expect the smallest distances","venue":null,"work_id":"fd1cdc75-3651-409e-a83c-c34a8cfd0146","year":null},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:11.119053Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:b28a92562b0f55b973d894636987f6c514c6b334c011023bcbfe664de285e97e","observation_id":"ce4bbb54-3138-4eec-b835-4404f612eb6d","resolution":{"observed_at":"2026-08-06T20:40:28.311331Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:28.052790Z","title":"This setting tests wh ether our metric can distinguish between speakers undistinguish - able with speech rate","venue":null,"work_id":"11888e26-bc00-4f1a-aee8-6ae2ba877ef8","year":null},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:11.225540Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:d1aa4b4175892c1bb40b8c45e09e13c90096c80234db6104174842871f17356b","observation_id":"8bbfef3e-0a69-429e-9162-8340ecfa3475","resolution":{"observed_at":"2026-08-06T20:40:28.102856Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:27.942508Z","title":"The results in the top section of Table 3 show that rhythm distances are signiﬁcantly larger between different speak ers, even those with similar speech rates","venue":null,"work_id":"c3837c28-f50f-4c6a-8e12-5ac167910874","year":null},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:11.368279Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:efb5a66c9c5f31d08816f66a6993db83258368d9251223c2b5d51b709445a8dc","observation_id":"89f98804-a7b9-49a1-99e0-d0bacc9f6e2c","resolution":{"observed_at":"2026-08-06T20:40:27.992700Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:27.831350Z","title":"We showed that ASV embeddings mainly encode speech identity markers relating to anatomy (e.g","venue":null,"work_id":"1128f45e-a179-4667-9367-4a0e54004a47","year":null},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:11.464887Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:62a7f432cb09b8a379c3c81259c9816836e8540c69166d701bf8184bfb5f7112","observation_id":"0ec6aae4-9475-4471-a80b-ff699a3f4477","resolution":{"observed_at":"2026-08-06T20:40:27.880864Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:27.675461Z","title":"An image is worth one word: Personalizing t ext-to- image generation using textual inversion,","venue":null,"work_id":"05960314-d5e2-42ab-a286-560553d518bc","year":2022},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:11.537350Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:a1eb1bfff4720224ec66306e65da0bb9c8a72c154e1d0030effc5e587c7fa11f","observation_id":"5d5d9667-5627-4180-8746-56411436c248","resolution":{"observed_at":"2026-08-06T20:40:27.756865Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:27.519522Z","title":"Dreambooth: Fine tuning text-to-image d iffusion models for subject-driven generation,","venue":null,"work_id":"cc2d1003-202c-40f9-b484-2b9365368e0d","year":2023},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:11.595403Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:59759629c33d2e2c3789c0ab5aebeff152a0e68e1625ebd085a4a4dcaa3e62f4","observation_id":"932e5017-ad30-4164-a120-a6d46a36daa7","resolution":{"observed_at":"2026-08-06T20:40:27.588974Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16771","last_updated":"2024-12-28T17:42:44Z","snapshot_observed_at":"2026-08-07T05:22:26.276397Z","submitted_at":"2024-04-25T17:23:43Z","title":"ConsistentID: Portrait Generation with Multimodal Fine-Grained Identity Preserving","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16771","snapshot_observed_at":"2026-08-06T20:40:11.679021Z","title":"ConsistentID: Portrait generation wit h multimodal ﬁne-grained identity preserving,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:11.679021Z"},"links":{"cited_paper":"/paper/2404.16771","citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:da6991931092f7476d1156ddb2d7baeae64380485f894713ea456ee364390dbd","observation_id":"0c62ff45-0144-4f18-aa42-7e8b96b9b4f1","resolution":{"observed_at":"2026-08-06T20:40:11.679021Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.15677","last_updated":"2024-04-27T14:24:15Z","snapshot_observed_at":"2026-07-06T18:04:50.648027Z","submitted_at":"2024-04-24T06:15:31Z","title":"CharacterFactory: Sampling Consistent Characters with GANs for Diffusion Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.15677","snapshot_observed_at":"2026-08-06T20:40:11.744848Z","title":"CharacterFactory: Sampling consistent charac- ters with gans for diffusion models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:11.744848Z"},"links":{"cited_paper":"/paper/2404.15677","citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:abc30cfd59331ced0c9848e2f34f7acc71b5d46176901d5205e5316b78f7a5e8","observation_id":"c89f5a9f-9e70-4ff9-a821-38f5722804a4","resolution":{"observed_at":"2026-08-06T20:40:11.744848Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:27.412634Z","title":"Generat ing video game scripts with style,","venue":null,"work_id":"cb7447de-ac9f-4aa1-bbd1-2d58e0c98fa1","year":2023},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:11.852395Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:ce5f588970c780f12b03b7c12c7c4dae2afeedb8f9033fc22d0fd960b4b79775","observation_id":"5d17b08b-2cef-42f0-b20c-065a1b807fe3","resolution":{"observed_at":"2026-08-06T20:40:27.458551Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:27.262995Z","title":"Meet your favorite character: Open-domai n chat- bot mimicking ﬁctional characters with only a few utterance s,","venue":null,"work_id":"d9de386f-80cf-4242-997f-9bbfa46fb784","year":2022},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:11.916503Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:b874109bf282f68ccfe1d5a753a11cce8f2ef83680e8ee455bcf7e2e219b1e8b","observation_id":"988925a3-6791-4cc9-b3ea-6d3176cae942","resolution":{"observed_at":"2026-08-06T20:40:27.351960Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:27.083772Z","title":"Information conveyed by vow- els","venue":null,"work_id":"cd96ee68-9402-47cf-9983-0569aece4875","year":1957},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:11.991064Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:3ccfeb119440d64f84ddc9cb44f217b77b0094a67cc1668d8a0f00f78e11fe3b","observation_id":"ba599d21-aa4a-44e9-9de5-a9c72b8282d5","resolution":{"observed_at":"2026-08-06T20:40:27.170164Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:26.898394Z","title":"The perception of personal iden tity in speech: Evidence from the perception of twins’ speech,","venue":null,"work_id":"9579b171-5488-4e81-902f-f44edbb212c0","year":null},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:12.078389Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:edd18e031193eb583f2956ee97c7a044ff536d8ab816293c49e21bd542b46e06","observation_id":"d6d6a64a-a140-4335-95bd-c7a4a8d02172","resolution":{"observed_at":"2026-08-06T20:40:26.987242Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:26.738901Z","title":"X-V ectors: Robust DNN embeddings for speaker recognition,","venue":null,"work_id":"6c851747-3ef1-438d-98c4-f8f1c3e4227c","year":2018},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:12.158681Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:64b17f932d70f067c64f2bb65417719edad1661d4e3df953c4738654bb2d7b91","observation_id":"e7c45ab1-d99e-4c5a-812e-3435e2d3b198","resolution":{"observed_at":"2026-08-06T20:40:26.806842Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:26.556175Z","title":"ECAPA - TDNN: emphasized channel attention, propagation and aggre ga- tion in TDNN based speaker veriﬁcation,","venue":null,"work_id":"788bb9d8-cbda-471c-bea4-b5c2f0064e4f","year":2020},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:12.245018Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:6ad2eb780a63f5c88b27d9a6bb1ea75df4cdb6c4229723d363583591611eb2f8","observation_id":"f7af280c-6e79-496f-a69c-c5d8babb5c99","resolution":{"observed_at":"2026-08-06T20:40:26.629537Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:26.383190Z","title":"State-of-the-art speaker recogni tion with neural network embeddings in NIST SRE18 and speakers in the wild evaluations,","venue":null,"work_id":"0dfb059a-c92a-4c8c-8493-5130801f6d8f","year":2020},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:12.319383Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:ec34cac3d071b462308a3dc651b4910d7f95b3dd731ddb221587147645e68b48","observation_id":"cc153765-41c9-49ce-b995-37e9c0f1de4d","resolution":{"observed_at":"2026-08-06T20:40:26.482722Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:26.230507Z","title":"Generalized end-to-end loss for speaker veriﬁca- tion,","venue":null,"work_id":"57e31d58-2a5a-4067-bb59-11b7ab69da16","year":2018},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:12.410015Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:076c45d9434330701a1df78a4b9566a454396ee29fbc77163472388de68a66d4","observation_id":"cb4977ac-8be1-44ea-ac18-7e54da19fefd","resolution":{"observed_at":"2026-08-06T20:40:26.302718Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2407.04291","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:17.234593Z","title":"We need variations in speech synthe- sis: Sub-center modelling for speaker embeddings,","venue":null,"work_id":"d5ccb00b-fd11-4c88-89ea-2807fa7d56e1","year":2024},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:12.504831Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:f8491238885cf937eb7778664c254f1628ed8d5dc5003b7abde29a2510ffeb09","observation_id":"b5ee12e4-44fe-4f82-9642-408fa023c43c","resolution":{"observed_at":"2026-08-06T20:40:17.374269Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:26.024786Z","title":"Predictions of subjective ratings and spooﬁng as- sessments of V oice Conversion Challenge 2020 submissions,","venue":null,"work_id":"c2d7b1fe-3377-48bd-9671-eb5e783f2f5d","year":2020},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:12.603592Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:eaa5a42b8ff53809c0539b7fb1efb05c7dc915168a11f75fdf9c7727ee872aca","observation_id":"7f5eebdb-5db5-4fd2-87f8-7537d1edc768","resolution":{"observed_at":"2026-08-06T20:40:26.121033Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:25.823833Z","title":"Transfer learning from speaker veriﬁcat ion to multi- speaker text-to-speech synthesis,","venue":null,"work_id":"230638e9-caf4-4ccb-ba34-5859b5bfc5af","year":2018},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:12.679830Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:cf230d05c71f30b8ad4fce9384801101bd5452f2b9880b124b4fedf51ae0d31c","observation_id":"b0d89eba-6eef-4933-97ed-e2d568c16f86","resolution":{"observed_at":"2026-08-06T20:40:25.902006Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:25.646386Z","title":"Zero-shot multi-speaker text-to-sp eech with state-of-the-art neural speaker embeddings,","venue":null,"work_id":"6160ada8-a738-4b37-9ff8-7cf96f989ca5","year":2020},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:12.740868Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:f51ea07cbea35a83da3387c841bb9df5a59952a1f9b87734cf2db6030f2e7774","observation_id":"d59ce44c-d9f4-412b-bedd-6b5094c18c70","resolution":{"observed_at":"2026-08-06T20:40:25.728737Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:25.519449Z","title":"Y ourTTS: towards zero-shot multi- speaker tts and zero-shot voice conversion for everyone,","venue":null,"work_id":"6f6a41dd-5f9e-442e-b94c-3ee95ce1c238","year":2022},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:12.837534Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:c63aa5d9163df24e2cbf70b9fcadc889cb1b29393485434b3841ce026d0b2ac3","observation_id":"38425e51-2504-4a84-874e-c042e926c1d6","resolution":{"observed_at":"2026-08-06T20:40:25.600183Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:25.362368Z","title":"Investigating on incorporating pret rained and learnable speaker representations for multi-speaker mult i-style text-to-speech,","venue":null,"work_id":"8a3ab763-88a4-4fa1-bc66-723eb2cb7016","year":2021},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:12.936260Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:5fde464d7dd38eab9297c2c6c6fcc04afe033463722e5e1ca04acb46549415d3","observation_id":"1cb6588a-a323-4f5c-b318-afc26a02eae4","resolution":{"observed_at":"2026-08-06T20:40:25.420748Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.05236","last_updated":"2025-07-22T21:32:13Z","snapshot_observed_at":"2026-08-08T21:47:45.872410Z","submitted_at":"2025-02-07T06:47:11Z","title":"Koel-TTS: Enhancing LLM based Speech Generation with Preference Alignment and Classifier Free Guidance","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.05236","snapshot_observed_at":"2026-08-06T20:40:13.032052Z","title":"Koel-TTS: Enhancing LLM based speec h gen- eration with preference alignment and classiﬁer free guida nce,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:13.032052Z"},"links":{"cited_paper":"/paper/2502.05236","citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:79c99ea36aedcffc7d0b958ce048a62ae18631effce5a8c985b794bd3564fcec","observation_id":"e3bf7555-52cb-4bf9-bd88-e292621223c0","resolution":{"observed_at":"2026-08-06T20:40:13.032052Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:25.126011Z","title":"The Multi-Speaker Multi-Style V oice Clo ning Chal- lenge 2021,","venue":null,"work_id":"ea14f8aa-c09e-4b7b-926e-bbbb58282780","year":2021},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:13.128895Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:017c31ef380512765353c75920d0be69315f81fe538affd67640c6e9c08b7ea1","observation_id":"01a602c7-7a3a-4992-bcca-2d20ea94f10d","resolution":{"observed_at":"2026-08-06T20:40:25.237363Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00529","last_updated":"2024-03-01T13:39:56Z","snapshot_observed_at":"2026-08-08T15:28:23.980784Z","submitted_at":"2024-03-01T13:39:56Z","title":"VoxGenesis: Unsupervised Discovery of Latent Speaker Manifold for Speech Synthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00529","snapshot_observed_at":"2026-08-06T20:40:13.209463Z","title":"V oxGenesis: Unsupervised discovery of latent speaker manifold for speech synthesis,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:13.209463Z"},"links":{"cited_paper":"/paper/2403.00529","citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:64fe6fcfc69de4270639008d8813c7e3314857b1d0e83e6144aa2cbafaebe216","observation_id":"4857260a-648a-4d2c-80a3-34fd08fb25de","resolution":{"observed_at":"2026-08-06T20:40:13.209463Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:24.943280Z","title":"Evaluating text-to-speech synthesis from a large discrete token-based speech language model,","venue":null,"work_id":"7a845f30-ce01-4651-bca5-9be72dc41bff","year":2024},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:13.330030Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:65cc502e946367c7d331a09cf6d41f44bdbc44656a5d5c9a3981ffb7457db6ae","observation_id":"fc46cf74-cdd9-420c-b7f8-d04b70d048b7","resolution":{"observed_at":"2026-08-06T20:40:25.020012Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:13.420536Z","title":"A comparison of discrete and soft speech units for improved voice conversion,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:13.420536Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:600f21661c90d787a7f311e5efe41c72a24607ef3246b0d5adad0996c276b2ba","observation_id":"f6e68c49-0f0f-4721-b3a1-898330ee7271","resolution":{"observed_at":"2026-08-06T20:40:13.420536Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:24.742265Z","title":"V oicebox: Text-guided multilingual univ ersal speech generation at scale,","venue":null,"work_id":"e7767a03-cfbc-4548-9014-dee1d2cf9114","year":2023},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:13.484801Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:175cfd8efc5b9bf2e5304a8bb27bb741bb777a57c59a9b0e63acae6ef57d1238","observation_id":"9fdce733-cd03-4761-b18d-ef7f05f91428","resolution":{"observed_at":"2026-08-06T20:40:24.864555Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:24.467120Z","title":"Neural codec language models are zero-s hot text to speech synthesizers,","venue":null,"work_id":"9be4b23c-efa6-4880-80dc-5c487ec26790","year":2025},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:13.544795Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:266b2d07cde1e0f8c7c69df9853513f7df6f43b7825bf20a0f2646b38536793a","observation_id":"67b8fbcc-5f6d-447d-bf18-4f153dc8ce3d","resolution":{"observed_at":"2026-08-06T20:40:24.567138Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:24.278278Z","title":"The Singing V oice Conversion Chall enge 2023,","venue":null,"work_id":"8eb32c08-65c6-4306-9239-3efd59cdcc73","year":2023},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:13.610042Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:d2823eee87300c4737ab714b95f946e9ba578d7db92560e9dd95c034114e3ca4","observation_id":"fc5d7913-3f76-4168-90e3-16b024875a87","resolution":{"observed_at":"2026-08-06T20:40:24.362172Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:24.054659Z","title":"Generative Data Augmentation Challe nge: Zero- shot speech synthesis for personalized speech enhancement ,","venue":null,"work_id":"0ff0424d-8bf8-49e4-b63d-284a3df2c414","year":2025},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:13.678518Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:2721003bb028fdca78db9a8003eac9c856ac58f74b9c35224722bd594a634edc","observation_id":"f3ef1777-0b8f-4ab0-9ddc-4580ee3c651c","resolution":{"observed_at":"2026-08-06T20:40:24.160692Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:23.819597Z","title":"Acoustic properties of voice timbre t ypes and their inﬂuence on voice classiﬁcation,","venue":null,"work_id":"0523ae2c-7bd6-493b-9b55-3d6cd2a556ba","year":1977},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:13.758589Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:8f3f4911b74af0e038942ab90d1f7a1f8e31ceda8381861c5eea994863e3afa1","observation_id":"654da8cc-1fdc-4b24-b46c-06147b004493","resolution":{"observed_at":"2026-08-06T20:40:23.920263Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:23.558301Z","title":"Speech production patterns in producing l inguis- tic contrasts are partly determined by individual differen ces in anatomy,","venue":null,"work_id":"728836a2-0f63-4b16-b0df-1a0b47f4de71","year":2018},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:13.823753Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:e61b3acb31ba7aeeaf7b8a1039c1bd621cd60d071d2cc68411c60fb89abdfed5","observation_id":"5c79d98a-4962-4f10-802e-f487f2340f69","resolution":{"observed_at":"2026-08-06T20:40:23.680062Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:23.377921Z","title":null,"venue":null,"work_id":"8ba5762d-51ee-4e27-90af-cdbdacd831d3","year":1980},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:13.871818Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:920757422c17351d8d18d0a9305098603027920e4b41718ef6871b7169d0b27d","observation_id":"bc66290e-98e4-4bd5-981a-3190fe79369b","resolution":{"observed_at":"2026-08-06T20:40:23.460801Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:23.181344Z","title":"A study of rhythm in London: Is syllable-timing a feature of Multicultural London English ?","venue":null,"work_id":"8b6d06d3-fd28-4d6b-bc35-49b741ef9009","year":2011},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:13.928173Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:ff6a0a3ad3b9b0890a6ad170ab163ea7e1f69cba97d8edb9e4c79e9275dc8b6e","observation_id":"cae77f0c-21b8-4b0b-8ad9-32e71334c683","resolution":{"observed_at":"2026-08-06T20:40:23.289528Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:22.942534Z","title":"The measurement of rhythm: a comparison of Sin- gapore and British English,","venue":null,"work_id":"4243b5e5-e913-4531-b666-d966cdd39a8f","year":2001},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:13.978443Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:8a427e08fbb6aa89d68329340f5de0369e2abc6da58bd0f1533a6b73d6d3af58","observation_id":"af5d4057-27d1-48f6-961b-011f8958c4ba","resolution":{"observed_at":"2026-08-06T20:40:23.067362Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:22.734150Z","title":"Sociophonetics of phonotactic pheno mena in french,","venue":null,"work_id":"d8df83cf-b1ee-4122-ae77-f1c66e639234","year":2015},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.042470Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:2818b0f814ad390a8e342a435dc77f6e87ed8da63fb1506474668d280c2abee6","observation_id":"10783b19-ca12-4bc5-8432-b0df7d8a26ba","resolution":{"observed_at":"2026-08-06T20:40:22.807797Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:22.514400Z","title":"Individual di fferences in vowel production,","venue":null,"work_id":"855efbe7-0dfe-49f8-9bd2-2cc75b3508e0","year":1993},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.108230Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:e533aae64eb7eb38b1b29c6c1a80f6efd2e4910dcf5864f19cbab31080c88aae","observation_id":"4d90a904-75f6-4cf3-8437-f4ba2f576871","resolution":{"observed_at":"2026-08-06T20:40:22.620821Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:22.332904Z","title":"Individual differences in speech pro- duction: V oice-onset-time,","venue":null,"work_id":"7b36e520-1d7a-44e6-8f59-e27d0018e0d3","year":2000},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.167152Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:d17ee51fa20659c16f43e51d0f0c067dd6c88fa7d2ba2dabdf23b19021513ce7","observation_id":"13627fac-ceff-4ca7-9de0-37f89d1c47df","resolution":{"observed_at":"2026-08-06T20:40:22.428908Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:22.136492Z","title":"The social life of phonetic s and phonology,","venue":null,"work_id":"0f1635b8-8bcb-4eba-a5e7-aed7147cf7ee","year":2006},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.225970Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:45148b064bd20466bd0baa8aa7f38837765f0c0495ef79afb9c67a3eb06d0581","observation_id":"6c2ed39d-60f7-4358-bb49-930065f56271","resolution":{"observed_at":"2026-08-06T20:40:22.226893Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:21.952085Z","title":"Prosodic rhythm and Afric an American english,","venue":null,"work_id":"3cd54b60-5db7-4cec-8e2f-73cdb31ae1fc","year":2006},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.273503Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:07f36ac865e2002726ccdc8b9bf8575933552e4fcdb04ef4ac459d86fc770fd1","observation_id":"a0a78c3e-a471-4b32-9051-7867fd9eefd9","resolution":{"observed_at":"2026-08-06T20:40:22.055134Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:21.800973Z","title":"La liaison sans enchaˆ ınement,","venue":null,"work_id":"63cba8b3-4a44-4a5f-99de-2574733fb0eb","year":1983},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.325252Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:fb7ef0f569dba61a193cbaa1267f73c8bdeccf767b58ae681d30681c3a78bb6c","observation_id":"dd9e3107-7a7a-413a-82ef-74ec4916a3e9","resolution":{"observed_at":"2026-08-06T20:40:21.865653Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:21.629519Z","title":"Stuart-Smith, E","venue":null,"work_id":"f87a97af-165a-4722-82a5-487909637f79","year":2014},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.381549Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:17e05e6049c96ff66623e931f1a29eaab6615a59982d5ed7943dbbf5fa5b9024","observation_id":"f7e19809-8f4a-47d6-b320-721f6b2d4a4b","resolution":{"observed_at":"2026-08-06T20:40:21.707914Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:21.418254Z","title":"The Blizzard Challenge 2023,","venue":null,"work_id":"818144ff-f5ae-4fde-bca0-93b633740eb2","year":2023},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.437994Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:dc748dfa947f4ab5844729802190f8c148a85f7e62f080ad14393155d61db96e","observation_id":"f2fee590-4f65-4a33-85bb-bb5f1f5ae99f","resolution":{"observed_at":"2026-08-06T20:40:21.492072Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:21.241899Z","title":"Reﬁning the evaluation of speech synthesis: A summ ary of the blizzard challenge 2023,","venue":null,"work_id":"5bca2b5b-7153-4047-a709-ca28b5881811","year":2023},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.486042Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:69289a773cfe66270e69a9f55a5df4c5ddc6b0266e66007cad506186214f0140","observation_id":"7c04876d-3c13-4c25-8171-85a9405c39fb","resolution":{"observed_at":"2026-08-06T20:40:21.310864Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:14.536826Z","title":"Good practices for evaluation of synthesized speech,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.536826Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:834b7d3ae1d4180f68295bb175b0223dc4311cf051a7be67eef2aed67ed34ae0","observation_id":"457a1f57-16e9-4294-83bd-5d1340707f4c","resolution":{"observed_at":"2026-08-06T20:40:14.536826Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:21.087978Z","title":"Stuck in the MOS pit: A critical anal ysis of MOS test methodology in TTS evaluation,","venue":null,"work_id":"8be43582-4194-4fb2-97f3-89fe20ca10b2","year":2023},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.599962Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:f9e3fccd58197ae03f3dd761cb34da8b23a0f3adaa88fac65246ed6ee9b2e46e","observation_id":"a8594ffb-9d77-4924-a681-932f2c5a1c65","resolution":{"observed_at":"2026-08-06T20:40:21.155580Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.12719","last_updated":"2025-05-27T02:40:41Z","snapshot_observed_at":"2026-08-03T19:15:50.931237Z","submitted_at":"2024-11-19T18:37:45Z","title":"Rethinking MUSHRA: Addressing Modern Challenges in Text-to-Speech Evaluation","version":3},"cited_work":{"arxiv_id":"2411.12719","doi":null,"metadata_source":"pith","pith_arxiv_id":"2411.12719","snapshot_observed_at":"2026-08-06T20:40:16.820041Z","title":"Rethinking MUSHRA: Addressing Modern Challenges in Text-to-Speech Evaluation","venue":"cs.CL","work_id":"ea42185f-ce9b-4c05-92df-4297fa0901f5","year":2024},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.655226Z"},"links":{"cited_paper":"/paper/2411.12719","citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:848d8478a866e40d61e1f82b31caf58878d0530a45365828cb2ff7886a53a5c3","observation_id":"bf0a2f81-a772-4739-9fd1-56a402a46427","resolution":{"observed_at":"2026-08-06T20:40:16.930605Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:20.926794Z","title":"MOS vs . AB: evaluating text-to-speech systems reliably using cluster ed stan- dard errors,","venue":null,"work_id":"ff955d57-e701-42c3-8b7f-f13a76faeae6","year":2023},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.709252Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:f25f61adafb243523e1558531660ede5b1cd1f3f36289a2cfbb4537ff4a3e506","observation_id":"26572b08-138c-4f2f-8822-ccf26c78472f","resolution":{"observed_at":"2026-08-06T20:40:21.015992Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:20.751152Z","title":"V oxSim: A perceptual voice similarity da taset,","venue":null,"work_id":"944f555f-f3d0-4433-8d47-f594651f96a7","year":2024},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.712118Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:8fab9b502712fbb5e11032ad9a1df94e8bc4336941a06835e59f8fbb1c109ab0","observation_id":"422c058a-9164-4776-a0b1-ed510877abc7","resolution":{"observed_at":"2026-08-06T20:40:20.835953Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:20.590042Z","title":"Salmon: A suite for acous tic language model evaluation,","venue":null,"work_id":"d1b5710d-afcf-486b-804f-1b5f9635b60a","year":2025},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.751685Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:2a6065d7ca3478d2cabc7a0714db4b6dd6239d4a218975e837484ff265a94546","observation_id":"e7fd25d3-fd91-4c7c-832e-14a34f7cd461","resolution":{"observed_at":"2026-08-06T20:40:20.663293Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:20.389992Z","title":"Automatic evaluation of speaker simila rity,","venue":null,"work_id":"5943ffae-c306-4802-b7af-dd6608ed46b9","year":2022},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.887907Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:709b2d8a3cb965752161c666e6d13030f58234688899d7ac657098a81ad5598e","observation_id":"82dfec55-c787-4bd5-b723-5017a25bfc68","resolution":{"observed_at":"2026-08-06T20:40:20.472116Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:20.219809Z","title":"SVSNet: An end-to-end speaker voice simi larity assessment model,","venue":null,"work_id":"edb80c0d-0be1-47f0-a563-0882fb11666b","year":2022},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:15.048606Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:e7a5b1d7a9a20f65b57a0dd899a894916c158f0fe4bdf31540bf931e70a2eea4","observation_id":"36a86cc9-028d-4d05-b97d-d2dfadd18548","resolution":{"observed_at":"2026-08-06T20:40:20.297397Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:20.058869Z","title":"The V oxCeleb Speaker Recognition Challe nge: a retrospective,","venue":null,"work_id":"10a41c4c-d83a-47f1-a280-6c6f4a88b4c3","year":2024},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:15.178145Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:a7589aaf54ed8a91ac10431e7d38f764a09a7bd4669d62c01dd816feebb1c39b","observation_id":"502ce3f8-0886-4a72-9e6a-87181406f5a4","resolution":{"observed_at":"2026-08-06T20:40:20.152485Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.04624","last_updated":"2021-06-08T18:22:56Z","snapshot_observed_at":"2026-08-04T07:20:14.379561Z","submitted_at":"2021-06-08T18:22:56Z","title":"SpeechBrain: A General-Purpose Speech Toolkit","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.04624","snapshot_observed_at":"2026-08-06T20:40:15.310440Z","title":"SpeechBrain: A General-Purpose Speech To olkit,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:15.310440Z"},"links":{"cited_paper":"/paper/2106.04624","citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:8d471c8560c58c47c369f0d1a4ef620e4b01a2c290964e398d4ebd6686b59ba2","observation_id":"d98022c9-0b21-4b97-a133-f6480c3df776","resolution":{"observed_at":"2026-08-06T20:40:15.310440Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:19.907921Z","title":"WavLM: large-scale self-supervised pr e-training for full stack speech processing,","venue":null,"work_id":"2ace4b71-aa6f-4f27-8d12-486c547ddb06","year":2022},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:15.439938Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:fd26d838155c6b3e840875febb62cd78ed3e4d257cdc21eda4fe1f60de6fb487","observation_id":"b2034870-5e78-47c0-8f37-9038336a86b6","resolution":{"observed_at":"2026-08-06T20:40:19.995774Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:19.758719Z","title":"The CMU Arctic speech databa ses,","venue":null,"work_id":"531837b8-b3a7-4258-a6d9-7e5e9a73a478","year":2004},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:15.547259Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:f2239a4690703ffd351dea4560b335eea08dd2314294766954fcb75594f041d5","observation_id":"65b8ea99-700f-4336-a54f-b31272b74d15","resolution":{"observed_at":"2026-08-06T20:40:19.842760Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:19.613645Z","title":"L2-ARCTIC: a non-native english speech c orpus,","venue":null,"work_id":"8bb117e6-a6c8-4809-913d-0644d63fac3e","year":2018},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:15.631101Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:2045455a68fbea6a5007a1f05c0314a314cd552fe9b6790ee564a5a4461214ac","observation_id":"56a9ff04-0491-43ae-bee3-ac8a523caf54","resolution":{"observed_at":"2026-08-06T20:40:19.681892Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:19.453657Z","title":"Lib- riSpeech: an ASR corpus based on public domain audio books,","venue":null,"work_id":"ff0afb23-39e6-4997-b955-6c24026a43ff","year":2015},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:15.701551Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:aaf7a8c783977f5c70f278c803e4d7bad1400f1525a671b79f7319e01d7b895d","observation_id":"33f54206-915c-4f9a-b19d-ddd85c8d66db","resolution":{"observed_at":"2026-08-06T20:40:19.533289Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:19.308751Z","title":"Analysis of fun- damental frequency, jitter, shimmer and vocal intensity in chil- dren with phonological disorders,","venue":null,"work_id":"f2ce5833-b451-4500-a1d2-651015b0efe0","year":2005},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:15.768394Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:f589797f1df7d0065523dc0ffb81f1276300a8306ddef15b870c092d2dec1478","observation_id":"e75cb6d1-b3e3-4a99-93e8-8046229d5370","resolution":{"observed_at":"2026-08-06T20:40:19.376682Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:19.125037Z","title":"V ocal acoust ic analy- sis – jitter, shimmer and HNR parameters,","venue":null,"work_id":"61e79a77-39d1-4a3d-8f00-90ec65458f7b","year":2013},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:15.849729Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:f3ff721b1c31e672a1efaa77906c63e4edfa9ae94727d9af7db6464bf2730f7c","observation_id":"af70a177-120a-4554-bcd0-0fe20b521ab6","resolution":{"observed_at":"2026-08-06T20:40:19.199832Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:18.942105Z","title":"V ariation of the acoustic parameters: f0, jitter, shimmer and alpha ratio in relation with different background noise levels,","venue":null,"work_id":"4edf17d8-d9bc-44c2-a37b-1699b9d8eaec","year":2023},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:15.945168Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:92014123a67e498cfc9e1d04f4113e8f876a4407c10dc65d648fb7cb50dd51d7","observation_id":"d80aa272-c9b8-486a-88f8-28b67beccaef","resolution":{"observed_at":"2026-08-06T20:40:19.053593Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:18.648723Z","title":"Rhyth m modeling for voice conversion,","venue":null,"work_id":"488e0a1a-546b-4d09-9296-c6a91e2bc285","year":2023},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:16.041293Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:68b593fa28e24203fa94c2000e32d84531ea5264b973ec4236356c8ff5c3202d","observation_id":"6bc0d3b7-7283-4f48-948d-5ea32a83db7b","resolution":{"observed_at":"2026-08-06T20:40:18.800943Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:18.324685Z","title":"Opensmile: the munich versatile and fast open-source audio feature extractor,","venue":null,"work_id":"f21da69c-a9bb-44b5-beb4-7e1e508fbd1b","year":2010},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:16.162090Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:d68bc8561a2a17d09e4e387d7a8d5987ef2711d1f3154c754634818fbca541f9","observation_id":"f993c02a-30c5-4bb9-832e-82ceaf2ffa39","resolution":{"observed_at":"2026-08-06T20:40:18.468730Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05421","last_updated":"2024-07-07T15:58:11Z","snapshot_observed_at":"2026-08-10T09:41:41.664674Z","submitted_at":"2024-07-07T15:58:11Z","title":"ASRRL-TTS: Agile Speaker Representation Reinforcement Learning for Text-to-Speech Speaker Adaptation","version":1},"cited_work":{"arxiv_id":"2407.05421","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.05421","snapshot_observed_at":"2026-08-06T20:40:16.626472Z","title":"ASRRL-TTS: Agile Speaker Representation Reinforcement Learning for Text-to-Speech Speaker Adaptation","venue":"eess.AS","work_id":"3d0f52c2-ca74-4a23-b1a2-041f5b18938c","year":2024},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:16.321959Z"},"links":{"cited_paper":"/paper/2407.05421","citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:52246fab38cdc1066139580f55b566f90079945eec33d6e13c0931231ca4810f","observation_id":"35d8cdf7-a228-477d-8e01-9ae7453031b1","resolution":{"observed_at":"2026-08-06T20:40:16.704034Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:18.025054Z","title":"All about audio equalization: Solu- tions and frontiers,","venue":null,"work_id":"7945530c-957f-4c2c-9098-140941ef77b1","year":2016},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:16.437239Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:c02c2f85a6cb69eeb9a3623629cb37d923dddcdef0e7c931ffff96b27c0a4601","observation_id":"fd0d3d8b-c6c7-42cb-afba-aeecf2eaf8f8","resolution":{"observed_at":"2026-08-06T20:40:18.192615Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-09T10:25:02.363099Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis"},"reference_resolution":{"displayed":69,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":8,"verified_exact":4,"verified_fuzzy":56},"total_outbound_references":69},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 69 of 69 outbound references and 1 inbound Pith citation observation for arXiv:2507.02176."}