{"as_of":"2026-08-19T04:08:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:28ef34c785db7f8f962dfe2f0d504aea256d9ea6c1fd8e256f1de940ed4fbe4e","coverage":[{"denominator":27,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":27,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T12:40:10.557940Z","state":"measured"},{"denominator":27,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":27,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2509.01391/citation-record","integrity":"/paper/2509.01391/integrity","json":"/paper/2509.01391/citation-record.json","paper":"/paper/2509.01391"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2203.01829","last_updated":"2022-03-01T11:15:35Z","snapshot_observed_at":"2026-08-16T17:17:51.073955Z","submitted_at":"2022-03-01T11:15:35Z","title":"A Brief Overview of Unsupervised Neural Speech Representation Learning","version":1},"cited_work":{"arxiv_id":"2203.01829","doi":null,"metadata_source":"pith","pith_arxiv_id":"2203.01829","snapshot_observed_at":"2026-08-05T12:40:10.673680Z","title":"A Brief Overview of Unsupervised Neural Speech Representation Learning","venue":"eess.AS","work_id":"6291de58-d3c7-4675-9c11-1ce25e18ce64","year":2022},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.427945Z"},"links":{"cited_paper":"/paper/2203.01829","citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:94fbd911520522af0930cd64024800991586a244626ec03533e125e1486506ac","observation_id":"5ff54f5e-682b-4d85-bd31-17cf031384a9","resolution":{"observed_at":"2026-08-05T12:40:10.681418Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:11.061493Z","title":"Natural TTS synthesis by conditioning Wavenet on mel-spectrogram predictions,","venue":null,"work_id":"a7a02e4e-1456-411b-b661-9f970627cad6","year":2018},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.433506Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:2eec218ee047c4e9dc8432a3087f7373c4f8d3c079be858ab6ef6f89450d8279","observation_id":"90f5f69d-626c-4cd5-ae28-15d4a175a960","resolution":{"observed_at":"2026-08-05T12:40:11.066460Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:11.044007Z","title":"Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech,","venue":null,"work_id":"3e22516d-71e2-406b-b20d-198c4d63a1a5","year":2021},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.439036Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:588195c3c0f72247a1d07db9e8efa2e1f52871f5e6f2a7958b354b56426ca362","observation_id":"8e44e2db-0d4c-446a-8314-cd30ad8b20d0","resolution":{"observed_at":"2026-08-05T12:40:11.049538Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:11.025029Z","title":"On generative spoken language modeling from raw audio,","venue":null,"work_id":"899d6ac9-1491-48c1-b0d6-2a9a541d94a3","year":2021},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.444464Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:11fd11fce6dc544ed54ae9011ee19dac90c24629c8ef7d3c30ce2122fab53083","observation_id":"9172d4dd-ad1a-421f-b1c9-3208c80ac016","resolution":{"observed_at":"2026-08-05T12:40:11.031437Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:11.008124Z","title":"Wav2vec 2.0: A framework for self-supervised learning of speech representations,","venue":null,"work_id":"bef77f70-2e96-40c2-9a9c-2f053f4a0c19","year":2020},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.450324Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:b92ce833a8ae032634dabc55ca81941176e45997a7bcce886bafc25aead26155","observation_id":"e34ba3e0-1027-4ac8-8070-ab1221fa4917","resolution":{"observed_at":"2026-08-05T12:40:11.013639Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.991441Z","title":"HuBERT: Self- supervised speech representation learning by masked prediction of hidden units,","venue":null,"work_id":"c8e78147-5e72-476a-88ed-80cc39364a29","year":2021},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.455036Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:a04e57ac8e6e8a876fceeab8a71a97a5723afb513b61ffccc6490379e523e180","observation_id":"c6abd5db-d1d3-4674-9079-3c8f01a02db8","resolution":{"observed_at":"2026-08-05T12:40:10.996936Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.974180Z","title":"A Unified Accent Estimation Method Based on Multi-Task Learn- ing for Japanese Text-to-Speech,","venue":null,"work_id":"26663653-70db-4d10-8bce-bbea5f72ea90","year":2022},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.460355Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:6b7b8b2c4d8879c30855ad44decbf5d0d0739d06e75a886dd7018c09c8004b7a","observation_id":"5e76a905-5d1d-4d2e-a1bb-11c2f2f79f8e","resolution":{"observed_at":"2026-08-05T12:40:10.979543Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.956871Z","title":"Reazonspeech: A free and massive corpus for Japanese ASR,","venue":null,"work_id":"9473f678-e372-4b89-bbb3-fb7e9196a88c","year":2023},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.465377Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:386d2ff277b2c5a56f6b9b3617685e987e96591a13f67c526551a9979a9491c5","observation_id":"c4c91880-fe62-47fb-9d30-c05813665721","resolution":{"observed_at":"2026-08-05T12:40:10.961640Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.938040Z","title":"Exploring the limits of transfer learning with a unified text-to- text transformer,","venue":null,"work_id":"890306a8-c90c-4458-8888-1d04c5fa09ae","year":2019},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.469772Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:447453fd2da31fa97272c56e6849bde36260cc53ddd7b87272e12fc58fbd0b42","observation_id":"daf8e8ad-0319-407d-99de-c4f775461830","resolution":{"observed_at":"2026-08-05T12:40:10.943901Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1711.00354","last_updated":"2017-10-28T05:28:01Z","snapshot_observed_at":"2026-08-14T20:19:26.882819Z","submitted_at":"2017-10-28T05:28:01Z","title":"JSUT corpus: free large-scale Japanese speech corpus for end-to-end speech synthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.00354","snapshot_observed_at":"2026-08-05T12:40:10.474344Z","title":"JSUT corpus: Free large-scale Japanese speech corpus for end- to-end speech synthesis,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.474344Z"},"links":{"cited_paper":"/paper/1711.00354","citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:d3964f7e7e2784c637e05fea67337961ef638eb3e849b1a8e7f510b500c46ed4","observation_id":"4326f94e-a02e-415e-a5c6-8d8e5a9fc790","resolution":{"observed_at":"2026-08-05T12:40:10.474344Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.06248","last_updated":"2019-08-17T06:04:46Z","snapshot_observed_at":"2026-08-17T09:27:25.572276Z","submitted_at":"2019-08-17T06:04:46Z","title":"JVS corpus: free Japanese multi-speaker voice corpus","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.06248","snapshot_observed_at":"2026-08-05T12:40:10.479287Z","title":"JVS corpus: Free Japanese multi- speaker voice corpus,","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.479287Z"},"links":{"cited_paper":"/paper/1908.06248","citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:4bf010779486da479110d68144e2b0b4bc0bfd548be4bd7d99a38396521dbedd","observation_id":"7237d5cf-fe98-43df-a014-99330b8e6193","resolution":{"observed_at":"2026-08-05T12:40:10.479287Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.921378Z","title":"Audio- book speech synthesis conditioned by cross-sentence context-aware word embeddings,","venue":null,"work_id":"f4099217-74f7-469d-bb78-5c84ce72c616","year":2021},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.484916Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:dce514897fc33099a04d86ac95960e21fa3986879952c519c7087849785a4649","observation_id":"02a76c6d-7964-40a0-8132-93a3e365dd55","resolution":{"observed_at":"2026-08-05T12:40:10.926464Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.903218Z","title":"J-MAC: Japanese multi-speaker audiobook corpus for speech synthesis,","venue":null,"work_id":"f8f642f9-3c96-4041-92bb-f4a4b0855c70","year":2022},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.489843Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:68b8eee299753f40edc81a1cc3276cb64802c0a16d9cb232f34e4b9dbe0ffc16","observation_id":"9d4ba3ca-6227-4aa6-8e29-535cc43dc83b","resolution":{"observed_at":"2026-08-05T12:40:10.908829Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.01793","last_updated":"2020-10-05T05:46:10Z","snapshot_observed_at":"2026-08-16T19:15:55.699345Z","submitted_at":"2020-10-05T05:46:10Z","title":"JSSS: free Japanese speech corpus for summarization and simplification","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.01793","snapshot_observed_at":"2026-08-05T12:40:10.494546Z","title":"JSSS: free japanese speech corpus for summarization and simplification,","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.494546Z"},"links":{"cited_paper":"/paper/2010.01793","citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:0bbe22c8d80e5eae5c105d8b3e5272b8a7eb064dc5e0148c1aafc831e8e618d9","observation_id":"f41b6dad-0f1d-4279-8df4-09aa58a892d9","resolution":{"observed_at":"2026-08-05T12:40:10.494546Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.885297Z","title":"T5g2p: Us- ing text-to-text transfer transformer for grapheme-to- phoneme conversion,","venue":null,"work_id":"0dd1280b-f10c-489d-b272-eda3ce566818","year":2021},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.500200Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:0a56f00a8e750991b8911159caf4fcc2d9d3ee7f99cce8492982c76a98707772","observation_id":"0d808ecc-efea-487b-9f18-88eb319e06ef","resolution":{"observed_at":"2026-08-05T12:40:10.890489Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.867472Z","title":"SpeechT5: Unified- modal encoder-decoder pre-training for spoken language processing,","venue":null,"work_id":"82fe7e8a-e655-4ff7-94cc-1437cdd4fe05","year":2022},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.504454Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:665df64d5aaebb676b3f0592863241552eb9a0e388d6c8bc8648954140196510","observation_id":"5af588d6-e2ec-44fe-819d-ace2c0e3a93e","resolution":{"observed_at":"2026-08-05T12:40:10.872317Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.851394Z","title":"Neural ma- chine translation for multilingual grapheme-to-phoneme conversion,","venue":null,"work_id":"5e7bdbf8-16de-4ca2-834e-cbb9d6bdf11c","year":2019},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.510410Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:522d9e30faa49c326c54ad63a282c74ca1d95ed6992025d7d9c9fbe34ba84fa2","observation_id":"36323435-2a5d-4930-81db-cf68a5d70b52","resolution":{"observed_at":"2026-08-05T12:40:10.856534Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.833643Z","title":"One model to pronounce them all: Multilingual grapheme- to-phoneme conversion with a transformer ensemble,","venue":null,"work_id":"6ecb89f6-0d67-41d3-8b25-7a6ffff0e0d2","year":2020},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.515754Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:6df63ce45196461c12cd3c39dbcdc608d193f6dc6cc426b8ebf2f74b9095feaf","observation_id":"13135d83-6c53-4f40-81dc-c563038ebeb5","resolution":{"observed_at":"2026-08-05T12:40:10.838454Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.815490Z","title":"Byt5 model for mas- sively multilingual grapheme-to-phoneme conversion,","venue":null,"work_id":"6180b7fb-b1d3-4d5c-8bd1-66707f640e00","year":2022},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.520427Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:43866976b1baeb3ba3de153687ea33aafc2848b807c9ba0c2a59d0191046599d","observation_id":"d50b1644-9ac2-43ee-bb72-12798424b365","resolution":{"observed_at":"2026-08-05T12:40:10.820105Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.799243Z","title":"ContentVec: An improved self-supervised speech representation by dis- entangling speakers,","venue":null,"work_id":"740361a7-3577-402a-8f5b-df726a6becc4","year":2022},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.524640Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:8cbc3a9f8ab01918679b171e42622a33f6ead7442bf4971aa5130e4c713c54e6","observation_id":"369ce2af-1ef8-40ed-84bf-32853dd86534","resolution":{"observed_at":"2026-08-05T12:40:10.803937Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-14T02:11:04.009144Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-05T12:40:10.529671Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.529671Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:c45a7b4d9d443e963eaa08c70fda87b87858699c7bbb57bbb0fc24d65d9f058a","observation_id":"1257031f-77b7-42fc-a4b6-f15c4431988c","resolution":{"observed_at":"2026-08-05T12:40:10.529671Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.780708Z","title":"Radford, K","venue":null,"work_id":"db1c4654-dc68-45c1-a362-483aa4bab2b3","year":2022},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.534763Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:5985c5f225ca8fe04de3e98bbbe63218c894819f13afa4f239ad4c1284f1e43e","observation_id":"b7c8b8da-ac01-4958-b309-6a285538d60a","resolution":{"observed_at":"2026-08-05T12:40:10.785436Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.763088Z","title":"UTMOS: UTokyo-SaruLab system for VoiceMOS challenge 2022,","venue":null,"work_id":"dcad5633-b255-4eee-bb13-8ad7d8b6226d","year":2022},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.539047Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:f6a8c112b8d0b333e608066edc8ac4f06e8a54b5a40490bdc63199cbdecf7045","observation_id":"e515bfbf-108a-4b89-929a-49b24a59a620","resolution":{"observed_at":"2026-08-05T12:40:10.768827Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.745533Z","title":"Speech quality assessment with W ARP-Q: From sim- ilarity to subsequence dynamic time warp cost,","venue":null,"work_id":"060e3f32-b69d-409a-867b-b4948bf90bfe","year":2022},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.543959Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:ca59463f94990b0d4eae99b81036b875bd506ea1063ad24304925ca00676eacb","observation_id":"2ae9a126-9e21-46af-a0d5-77bf7ccb7e17","resolution":{"observed_at":"2026-08-05T12:40:10.750702Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.726743Z","title":"Per- ceptual evaluation of speech quality (PESQ)-a new method for speech quality assessment of telephone networks and codecs,","venue":null,"work_id":"7d241002-8a56-4a2a-bb8b-fd698a476cd3","year":2001},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.548721Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:fe04a67edb2c28a25779f7b6b2a63ae4133352152c9f2087f73d26ae71d20b93","observation_id":"90e04134-54e2-4c17-8272-80191bbcc018","resolution":{"observed_at":"2026-08-05T12:40:10.732694Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.708663Z","title":"mT5: A mas- sively multilingual pre-trained text-to-text transformer,","venue":null,"work_id":"8be60b7b-b938-4452-9d33-c59929d6a86c","year":2021},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.553662Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:b21f300c935cfaac03900ae5178f84d8e59347bd5d1d08663fe1512ac14301f0","observation_id":"07250b96-e2a1-4327-9ce1-90001fc03b31","resolution":{"observed_at":"2026-08-05T12:40:10.713327Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.693256Z","title":"ByT5: Towards a token-free future with pre-trained byte-to-byte models,","venue":null,"work_id":"aa553116-8af1-4b5e-99da-778a87de4f1a","year":2022},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.557940Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:3e63ae33a7e7fc6a8d5043485dfc1ebab9437f8b2c3ec10a966f70ad77fd67f0","observation_id":"d9cf1ded-3c45-4a9f-957d-e7bd01a8cb69","resolution":{"observed_at":"2026-08-05T12:40:10.697841Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","latest_version":1,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model"},"reference_resolution":{"displayed":27,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":4,"verified_exact":1,"verified_fuzzy":22},"total_outbound_references":27},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 19 August 2026, this Paper Citation Record lists 27 of 27 outbound references and 0 inbound Pith citation observations for arXiv:2509.01391."}