{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FGRYNMAHWGAYZXG5R4VPMAATFX","short_pith_number":"pith:FGRYNMAH","schema_version":"1.0","canonical_sha256":"29a386b007b1818cdcdd8f2af600132dff88c5f171481b6a150cf959c7b64b7f","source":{"kind":"arxiv","id":"2401.05614","version":1},"attestation_state":"computed","paper":{"title":"Self-Attention and Hybrid Features for Replay and Deep-Fake Audio Detection","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.MM","eess.AS"],"primary_cat":"cs.SD","authors_text":"Chi-Man Pun, Lian Huang","submitted_at":"2024-01-11T01:41:16Z","abstract_excerpt":"Due to the successful application of deep learning, audio spoofing detection has made significant progress. Spoofed audio with speech synthesis or voice conversion can be well detected by many countermeasures. However, an automatic speaker verification system is still vulnerable to spoofing attacks such as replay or Deep-Fake audio. Deep-Fake audio means that the spoofed utterances are generated using text-to-speech (TTS) and voice conversion (VC) algorithms. Here, we propose a novel framework based on hybrid features with the self-attention mechanism. It is expected that hybrid features can b"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.05614","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.SD","submitted_at":"2024-01-11T01:41:16Z","cross_cats_sorted":["cs.MM","eess.AS"],"title_canon_sha256":"2af6f951f439c8bf836fbd2084a625e965a71267917da205db2d14f11f45fd34","abstract_canon_sha256":"1c56fb072826b5ccbc827f5e6b631c21457ad583cacb3e66a49c108c88171b3f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:32:32.599797Z","signature_b64":"HSwBBI7+/AWxdHfOxn2wG5XPDz570Ez3eKUev3jjmZcp12XLpyLDEgjpxTFwdBL1BJJY62rSB7eJIbYsrQXAAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"29a386b007b1818cdcdd8f2af600132dff88c5f171481b6a150cf959c7b64b7f","last_reissued_at":"2026-07-05T07:32:32.599306Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:32:32.599306Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Self-Attention and Hybrid Features for Replay and Deep-Fake Audio Detection","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.MM","eess.AS"],"primary_cat":"cs.SD","authors_text":"Chi-Man Pun, Lian Huang","submitted_at":"2024-01-11T01:41:16Z","abstract_excerpt":"Due to the successful application of deep learning, audio spoofing detection has made significant progress. Spoofed audio with speech synthesis or voice conversion can be well detected by many countermeasures. However, an automatic speaker verification system is still vulnerable to spoofing attacks such as replay or Deep-Fake audio. Deep-Fake audio means that the spoofed utterances are generated using text-to-speech (TTS) and voice conversion (VC) algorithms. Here, we propose a novel framework based on hybrid features with the self-attention mechanism. It is expected that hybrid features can b"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.05614","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.05614/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.05614","created_at":"2026-07-05T07:32:32.599370+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.05614v1","created_at":"2026-07-05T07:32:32.599370+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.05614","created_at":"2026-07-05T07:32:32.599370+00:00"},{"alias_kind":"pith_short_12","alias_value":"FGRYNMAHWGAY","created_at":"2026-07-05T07:32:32.599370+00:00"},{"alias_kind":"pith_short_16","alias_value":"FGRYNMAHWGAYZXG5","created_at":"2026-07-05T07:32:32.599370+00:00"},{"alias_kind":"pith_short_8","alias_value":"FGRYNMAH","created_at":"2026-07-05T07:32:32.599370+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2411.19841","citing_title":"Parallel Stacked Aggregated Network for Voice Authentication in IoT-Enabled Smart Devices","ref_index":15,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FGRYNMAHWGAYZXG5R4VPMAATFX","json":"https://pith.science/pith/FGRYNMAHWGAYZXG5R4VPMAATFX.json","graph_json":"https://pith.science/api/pith-number/FGRYNMAHWGAYZXG5R4VPMAATFX/graph.json","events_json":"https://pith.science/api/pith-number/FGRYNMAHWGAYZXG5R4VPMAATFX/events.json","paper":"https://pith.science/paper/FGRYNMAH"},"agent_actions":{"view_html":"https://pith.science/pith/FGRYNMAHWGAYZXG5R4VPMAATFX","download_json":"https://pith.science/pith/FGRYNMAHWGAYZXG5R4VPMAATFX.json","view_paper":"https://pith.science/paper/FGRYNMAH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.05614&json=true","fetch_graph":"https://pith.science/api/pith-number/FGRYNMAHWGAYZXG5R4VPMAATFX/graph.json","fetch_events":"https://pith.science/api/pith-number/FGRYNMAHWGAYZXG5R4VPMAATFX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FGRYNMAHWGAYZXG5R4VPMAATFX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FGRYNMAHWGAYZXG5R4VPMAATFX/action/storage_attestation","attest_author":"https://pith.science/pith/FGRYNMAHWGAYZXG5R4VPMAATFX/action/author_attestation","sign_citation":"https://pith.science/pith/FGRYNMAHWGAYZXG5R4VPMAATFX/action/citation_signature","submit_replication":"https://pith.science/pith/FGRYNMAHWGAYZXG5R4VPMAATFX/action/replication_record"}},"created_at":"2026-07-05T07:32:32.599370+00:00","updated_at":"2026-07-05T07:32:32.599370+00:00"}