{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:NEG2JMP5UWYQLOKQ7ZLSQQCMDF","short_pith_number":"pith:NEG2JMP5","schema_version":"1.0","canonical_sha256":"690da4b1fda5b105b950fe5728404c19445a847e961508076277489dd9833dcd","source":{"kind":"arxiv","id":"2203.04114","version":3},"attestation_state":"computed","paper":{"title":"A study on joint modeling and data augmentation of multi-modalities for audio-visual scene classification","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.SD","eess.AS"],"primary_cat":"cs.MM","authors_text":"Chao-Han Huck Yang, Chin-Hui Lee, Hu Hu, Jun Du, Qing Wang, Sabato Marco Siniscalchi, Siyuan Zheng, Yajian Wang, Yannan Wang, Yunqing Li, Yuzhong Wu","submitted_at":"2022-03-07T07:29:55Z","abstract_excerpt":"In this paper, we propose two techniques, namely joint modeling and data augmentation, to improve system performances for audio-visual scene classification (AVSC). We employ pre-trained networks trained only on image data sets to extract video embedding; whereas for audio embedding models, we decide to train them from scratch. We explore different neural network architectures for joint modeling to effectively combine the video and audio modalities. Moreover, data augmentation strategies are investigated to increase audio-visual training set size. For the video modality the effectiveness of sev"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2203.04114","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.MM","submitted_at":"2022-03-07T07:29:55Z","cross_cats_sorted":["cs.CV","cs.SD","eess.AS"],"title_canon_sha256":"6b45bdf6e90aa911c27a9e2af1fc4242b383ee0d880d17124de4c0b2cc853a1a","abstract_canon_sha256":"59df944c30521658857ca827accd421e45b48f1c163e6a661c5f882edb087e91"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:53:41.246291Z","signature_b64":"MhHFQF/c2fI1GNTlXTZzk6dSPRpoVeDmFLkt1cluv2vW8372ddootX5Ihku+z69Fzb+cPGI0kaikDPNOyv8wAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"690da4b1fda5b105b950fe5728404c19445a847e961508076277489dd9833dcd","last_reissued_at":"2026-07-05T04:53:41.245872Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:53:41.245872Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A study on joint modeling and data augmentation of multi-modalities for audio-visual scene classification","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.SD","eess.AS"],"primary_cat":"cs.MM","authors_text":"Chao-Han Huck Yang, Chin-Hui Lee, Hu Hu, Jun Du, Qing Wang, Sabato Marco Siniscalchi, Siyuan Zheng, Yajian Wang, Yannan Wang, Yunqing Li, Yuzhong Wu","submitted_at":"2022-03-07T07:29:55Z","abstract_excerpt":"In this paper, we propose two techniques, namely joint modeling and data augmentation, to improve system performances for audio-visual scene classification (AVSC). We employ pre-trained networks trained only on image data sets to extract video embedding; whereas for audio embedding models, we decide to train them from scratch. We explore different neural network architectures for joint modeling to effectively combine the video and audio modalities. Moreover, data augmentation strategies are investigated to increase audio-visual training set size. For the video modality the effectiveness of sev"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.04114","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.04114/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2203.04114","created_at":"2026-07-05T04:53:41.245929+00:00"},{"alias_kind":"arxiv_version","alias_value":"2203.04114v3","created_at":"2026-07-05T04:53:41.245929+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.04114","created_at":"2026-07-05T04:53:41.245929+00:00"},{"alias_kind":"pith_short_12","alias_value":"NEG2JMP5UWYQ","created_at":"2026-07-05T04:53:41.245929+00:00"},{"alias_kind":"pith_short_16","alias_value":"NEG2JMP5UWYQLOKQ","created_at":"2026-07-05T04:53:41.245929+00:00"},{"alias_kind":"pith_short_8","alias_value":"NEG2JMP5","created_at":"2026-07-05T04:53:41.245929+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NEG2JMP5UWYQLOKQ7ZLSQQCMDF","json":"https://pith.science/pith/NEG2JMP5UWYQLOKQ7ZLSQQCMDF.json","graph_json":"https://pith.science/api/pith-number/NEG2JMP5UWYQLOKQ7ZLSQQCMDF/graph.json","events_json":"https://pith.science/api/pith-number/NEG2JMP5UWYQLOKQ7ZLSQQCMDF/events.json","paper":"https://pith.science/paper/NEG2JMP5"},"agent_actions":{"view_html":"https://pith.science/pith/NEG2JMP5UWYQLOKQ7ZLSQQCMDF","download_json":"https://pith.science/pith/NEG2JMP5UWYQLOKQ7ZLSQQCMDF.json","view_paper":"https://pith.science/paper/NEG2JMP5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2203.04114&json=true","fetch_graph":"https://pith.science/api/pith-number/NEG2JMP5UWYQLOKQ7ZLSQQCMDF/graph.json","fetch_events":"https://pith.science/api/pith-number/NEG2JMP5UWYQLOKQ7ZLSQQCMDF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NEG2JMP5UWYQLOKQ7ZLSQQCMDF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NEG2JMP5UWYQLOKQ7ZLSQQCMDF/action/storage_attestation","attest_author":"https://pith.science/pith/NEG2JMP5UWYQLOKQ7ZLSQQCMDF/action/author_attestation","sign_citation":"https://pith.science/pith/NEG2JMP5UWYQLOKQ7ZLSQQCMDF/action/citation_signature","submit_replication":"https://pith.science/pith/NEG2JMP5UWYQLOKQ7ZLSQQCMDF/action/replication_record"}},"created_at":"2026-07-05T04:53:41.245929+00:00","updated_at":"2026-07-05T04:53:41.245929+00:00"}