{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6ZVNCA6QWVU7IOGMA3GFKDXIDU","short_pith_number":"pith:6ZVNCA6Q","schema_version":"1.0","canonical_sha256":"f66ad103d0b569f438cc06cc550ee81d3458eee57a23f983dbda2b7e02c1de73","source":{"kind":"arxiv","id":"2412.16771","version":1},"attestation_state":"computed","paper":{"title":"SilVar: Speech Driven Multimodal Model for Reasoning Visual Question Answering and Object Localization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chris Ngo, Hoang-Nam Le, Phu-Vinh Nguyen, Tan-Hanh Pham, Truong-Son Hy","submitted_at":"2024-12-21T20:52:32Z","abstract_excerpt":"Visual Language Models have demonstrated remarkable capabilities across tasks, including visual question answering and image captioning. However, most models rely on text-based instructions, limiting their effectiveness in human-machine interactions. Moreover, the quality of language models depends on reasoning and prompting techniques, such as COT, which remain underexplored when using speech instructions. To address these challenges, we propose SilVar, a novel end-to-end multimodal model that uses speech instructions for reasoning in visual question answering. In addition, we investigate rea"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.16771","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-12-21T20:52:32Z","cross_cats_sorted":[],"title_canon_sha256":"042dbd50956bf2999c3d8380281f51c5ed1c71a3b6926bab921e9fa3f12dd8ce","abstract_canon_sha256":"05702a0784066018882f906d7ff9795ddcf14def022eea27e7c890b86905bddf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:53:12.880528Z","signature_b64":"yyPcPdTtbNOy2LxDqVh7jeH0pMbQp1C30OkY9Aa9qrR3U0+8moF5w0KvgzmfBvN7EaUVh10Jkoh5oz7PCI/FAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f66ad103d0b569f438cc06cc550ee81d3458eee57a23f983dbda2b7e02c1de73","last_reissued_at":"2026-07-05T09:53:12.880072Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:53:12.880072Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SilVar: Speech Driven Multimodal Model for Reasoning Visual Question Answering and Object Localization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chris Ngo, Hoang-Nam Le, Phu-Vinh Nguyen, Tan-Hanh Pham, Truong-Son Hy","submitted_at":"2024-12-21T20:52:32Z","abstract_excerpt":"Visual Language Models have demonstrated remarkable capabilities across tasks, including visual question answering and image captioning. However, most models rely on text-based instructions, limiting their effectiveness in human-machine interactions. Moreover, the quality of language models depends on reasoning and prompting techniques, such as COT, which remain underexplored when using speech instructions. To address these challenges, we propose SilVar, a novel end-to-end multimodal model that uses speech instructions for reasoning in visual question answering. In addition, we investigate rea"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.16771","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.16771/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.16771","created_at":"2026-07-05T09:53:12.880133+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.16771v1","created_at":"2026-07-05T09:53:12.880133+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.16771","created_at":"2026-07-05T09:53:12.880133+00:00"},{"alias_kind":"pith_short_12","alias_value":"6ZVNCA6QWVU7","created_at":"2026-07-05T09:53:12.880133+00:00"},{"alias_kind":"pith_short_16","alias_value":"6ZVNCA6QWVU7IOGM","created_at":"2026-07-05T09:53:12.880133+00:00"},{"alias_kind":"pith_short_8","alias_value":"6ZVNCA6Q","created_at":"2026-07-05T09:53:12.880133+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6ZVNCA6QWVU7IOGMA3GFKDXIDU","json":"https://pith.science/pith/6ZVNCA6QWVU7IOGMA3GFKDXIDU.json","graph_json":"https://pith.science/api/pith-number/6ZVNCA6QWVU7IOGMA3GFKDXIDU/graph.json","events_json":"https://pith.science/api/pith-number/6ZVNCA6QWVU7IOGMA3GFKDXIDU/events.json","paper":"https://pith.science/paper/6ZVNCA6Q"},"agent_actions":{"view_html":"https://pith.science/pith/6ZVNCA6QWVU7IOGMA3GFKDXIDU","download_json":"https://pith.science/pith/6ZVNCA6QWVU7IOGMA3GFKDXIDU.json","view_paper":"https://pith.science/paper/6ZVNCA6Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.16771&json=true","fetch_graph":"https://pith.science/api/pith-number/6ZVNCA6QWVU7IOGMA3GFKDXIDU/graph.json","fetch_events":"https://pith.science/api/pith-number/6ZVNCA6QWVU7IOGMA3GFKDXIDU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6ZVNCA6QWVU7IOGMA3GFKDXIDU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6ZVNCA6QWVU7IOGMA3GFKDXIDU/action/storage_attestation","attest_author":"https://pith.science/pith/6ZVNCA6QWVU7IOGMA3GFKDXIDU/action/author_attestation","sign_citation":"https://pith.science/pith/6ZVNCA6QWVU7IOGMA3GFKDXIDU/action/citation_signature","submit_replication":"https://pith.science/pith/6ZVNCA6QWVU7IOGMA3GFKDXIDU/action/replication_record"}},"created_at":"2026-07-05T09:53:12.880133+00:00","updated_at":"2026-07-05T09:53:12.880133+00:00"}