{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:6OCJHCOM6375DBECHK64VTZ3M5","short_pith_number":"pith:6OCJHCOM","schema_version":"1.0","canonical_sha256":"f3849389ccf6ffd184823abdcacf3b675866a84003ec56be06da5032156aa1ad","source":{"kind":"arxiv","id":"2305.15403","version":1},"attestation_state":"computed","paper":{"title":"AV-TranSpeech: Audio-Visual Robust Speech-to-Speech Translation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Huadai Liu, Jinglin Liu, Jinzheng He, Lichao Zhang, Linjun Li, Rongjie Huang, Xiang Yin, Xize Cheng, Yi Ren, Zhenhui Ye, Zhou Zhao","submitted_at":"2023-05-24T17:59:03Z","abstract_excerpt":"Direct speech-to-speech translation (S2ST) aims to convert speech from one language into another, and has demonstrated significant progress to date. Despite the recent success, current S2ST models still suffer from distinct degradation in noisy environments and fail to translate visual speech (i.e., the movement of lips and teeth). In this work, we present AV-TranSpeech, the first audio-visual speech-to-speech (AV-S2ST) translation model without relying on intermediate text. AV-TranSpeech complements the audio stream with visual information to promote system robustness and opens up a host of p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.15403","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-05-24T17:59:03Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"b114003d23bf0ddf8e496a44d9009265cc1dc548a8477167097f942edf1307af","abstract_canon_sha256":"24d896ee6b622e657c786b3284342548febfbf2b98d527c3954ff3a61222b1d3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:13:34.864358Z","signature_b64":"JhDbweZnE+SeAlX8L8gS2SpLviL8kCVSusoh73z1n0yKuc4FjwJUtEWpooH4pnxUpBcbS1mD7WFY3xb02wlNBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f3849389ccf6ffd184823abdcacf3b675866a84003ec56be06da5032156aa1ad","last_reissued_at":"2026-07-05T06:13:34.863930Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:13:34.863930Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AV-TranSpeech: Audio-Visual Robust Speech-to-Speech Translation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Huadai Liu, Jinglin Liu, Jinzheng He, Lichao Zhang, Linjun Li, Rongjie Huang, Xiang Yin, Xize Cheng, Yi Ren, Zhenhui Ye, Zhou Zhao","submitted_at":"2023-05-24T17:59:03Z","abstract_excerpt":"Direct speech-to-speech translation (S2ST) aims to convert speech from one language into another, and has demonstrated significant progress to date. Despite the recent success, current S2ST models still suffer from distinct degradation in noisy environments and fail to translate visual speech (i.e., the movement of lips and teeth). In this work, we present AV-TranSpeech, the first audio-visual speech-to-speech (AV-S2ST) translation model without relying on intermediate text. AV-TranSpeech complements the audio stream with visual information to promote system robustness and opens up a host of p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.15403","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.15403/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.15403","created_at":"2026-07-05T06:13:34.863985+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.15403v1","created_at":"2026-07-05T06:13:34.863985+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.15403","created_at":"2026-07-05T06:13:34.863985+00:00"},{"alias_kind":"pith_short_12","alias_value":"6OCJHCOM6375","created_at":"2026-07-05T06:13:34.863985+00:00"},{"alias_kind":"pith_short_16","alias_value":"6OCJHCOM6375DBEC","created_at":"2026-07-05T06:13:34.863985+00:00"},{"alias_kind":"pith_short_8","alias_value":"6OCJHCOM","created_at":"2026-07-05T06:13:34.863985+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.16530","citing_title":"Improving Lip-synchrony in Direct Audio-Visual Speech-to-Speech Translation","ref_index":21,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6OCJHCOM6375DBECHK64VTZ3M5","json":"https://pith.science/pith/6OCJHCOM6375DBECHK64VTZ3M5.json","graph_json":"https://pith.science/api/pith-number/6OCJHCOM6375DBECHK64VTZ3M5/graph.json","events_json":"https://pith.science/api/pith-number/6OCJHCOM6375DBECHK64VTZ3M5/events.json","paper":"https://pith.science/paper/6OCJHCOM"},"agent_actions":{"view_html":"https://pith.science/pith/6OCJHCOM6375DBECHK64VTZ3M5","download_json":"https://pith.science/pith/6OCJHCOM6375DBECHK64VTZ3M5.json","view_paper":"https://pith.science/paper/6OCJHCOM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.15403&json=true","fetch_graph":"https://pith.science/api/pith-number/6OCJHCOM6375DBECHK64VTZ3M5/graph.json","fetch_events":"https://pith.science/api/pith-number/6OCJHCOM6375DBECHK64VTZ3M5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6OCJHCOM6375DBECHK64VTZ3M5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6OCJHCOM6375DBECHK64VTZ3M5/action/storage_attestation","attest_author":"https://pith.science/pith/6OCJHCOM6375DBECHK64VTZ3M5/action/author_attestation","sign_citation":"https://pith.science/pith/6OCJHCOM6375DBECHK64VTZ3M5/action/citation_signature","submit_replication":"https://pith.science/pith/6OCJHCOM6375DBECHK64VTZ3M5/action/replication_record"}},"created_at":"2026-07-05T06:13:34.863985+00:00","updated_at":"2026-07-05T06:13:34.863985+00:00"}