{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:SDQB44YAHFNTS2VG3CVVLBJ67O","short_pith_number":"pith:SDQB44YA","schema_version":"1.0","canonical_sha256":"90e01e7300395b396aa6d8ab55853efb8eda76c5e5e1e5245605ecadb7ad6261","source":{"kind":"arxiv","id":"2310.04982","version":1},"attestation_state":"computed","paper":{"title":"Comparative Analysis of Transfer Learning in Deep Learning Text-to-Speech Models on a Few-Shot, Low-Resource, Customized Dataset","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","eess.AS"],"primary_cat":"cs.SD","authors_text":"Ze Liu","submitted_at":"2023-10-08T03:08:25Z","abstract_excerpt":"Text-to-Speech (TTS) synthesis using deep learning relies on voice quality. Modern TTS models are advanced, but they need large amount of data. Given the growing computational complexity of these models and the scarcity of large, high-quality datasets, this research focuses on transfer learning, especially on few-shot, low-resource, and customized datasets. In this research, \"low-resource\" specifically refers to situations where there are limited amounts of training data, such as a small number of audio recordings and corresponding transcriptions for a particular language or dialect. This thes"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.04982","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.SD","submitted_at":"2023-10-08T03:08:25Z","cross_cats_sorted":["cs.AI","cs.CL","eess.AS"],"title_canon_sha256":"e7387070784ec3179b58dd1a7e466b9549f62f757b4900290d6056bd5bd7e7f2","abstract_canon_sha256":"44a638fea6ccbe94939c3ba6da6ea05bb2aea94a147f2db8986d0198c580b27c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:58:25.228485Z","signature_b64":"tt7n90cUAaE/L2/EAQCCMwirftlCZsIeULWMsoVBs8nrDNCdtQAaKZVtq0P5hczqE2PmcX2spCNbPLvwvshpBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"90e01e7300395b396aa6d8ab55853efb8eda76c5e5e1e5245605ecadb7ad6261","last_reissued_at":"2026-07-05T06:58:25.227981Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:58:25.227981Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Comparative Analysis of Transfer Learning in Deep Learning Text-to-Speech Models on a Few-Shot, Low-Resource, Customized Dataset","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","eess.AS"],"primary_cat":"cs.SD","authors_text":"Ze Liu","submitted_at":"2023-10-08T03:08:25Z","abstract_excerpt":"Text-to-Speech (TTS) synthesis using deep learning relies on voice quality. Modern TTS models are advanced, but they need large amount of data. Given the growing computational complexity of these models and the scarcity of large, high-quality datasets, this research focuses on transfer learning, especially on few-shot, low-resource, and customized datasets. In this research, \"low-resource\" specifically refers to situations where there are limited amounts of training data, such as a small number of audio recordings and corresponding transcriptions for a particular language or dialect. This thes"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.04982","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.04982/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.04982","created_at":"2026-07-05T06:58:25.228043+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.04982v1","created_at":"2026-07-05T06:58:25.228043+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.04982","created_at":"2026-07-05T06:58:25.228043+00:00"},{"alias_kind":"pith_short_12","alias_value":"SDQB44YAHFNT","created_at":"2026-07-05T06:58:25.228043+00:00"},{"alias_kind":"pith_short_16","alias_value":"SDQB44YAHFNTS2VG","created_at":"2026-07-05T06:58:25.228043+00:00"},{"alias_kind":"pith_short_8","alias_value":"SDQB44YA","created_at":"2026-07-05T06:58:25.228043+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.10885","citing_title":"BanglaFake: Constructing and Evaluating a Specialized Bengali Deepfake Audio Dataset","ref_index":19,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SDQB44YAHFNTS2VG3CVVLBJ67O","json":"https://pith.science/pith/SDQB44YAHFNTS2VG3CVVLBJ67O.json","graph_json":"https://pith.science/api/pith-number/SDQB44YAHFNTS2VG3CVVLBJ67O/graph.json","events_json":"https://pith.science/api/pith-number/SDQB44YAHFNTS2VG3CVVLBJ67O/events.json","paper":"https://pith.science/paper/SDQB44YA"},"agent_actions":{"view_html":"https://pith.science/pith/SDQB44YAHFNTS2VG3CVVLBJ67O","download_json":"https://pith.science/pith/SDQB44YAHFNTS2VG3CVVLBJ67O.json","view_paper":"https://pith.science/paper/SDQB44YA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.04982&json=true","fetch_graph":"https://pith.science/api/pith-number/SDQB44YAHFNTS2VG3CVVLBJ67O/graph.json","fetch_events":"https://pith.science/api/pith-number/SDQB44YAHFNTS2VG3CVVLBJ67O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SDQB44YAHFNTS2VG3CVVLBJ67O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SDQB44YAHFNTS2VG3CVVLBJ67O/action/storage_attestation","attest_author":"https://pith.science/pith/SDQB44YAHFNTS2VG3CVVLBJ67O/action/author_attestation","sign_citation":"https://pith.science/pith/SDQB44YAHFNTS2VG3CVVLBJ67O/action/citation_signature","submit_replication":"https://pith.science/pith/SDQB44YAHFNTS2VG3CVVLBJ67O/action/replication_record"}},"created_at":"2026-07-05T06:58:25.228043+00:00","updated_at":"2026-07-05T06:58:25.228043+00:00"}