{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:OLFDN42DCUOODYC5ZAMOMXPIPZ","short_pith_number":"pith:OLFDN42D","schema_version":"1.0","canonical_sha256":"72ca36f343151ce1e05dc818e65de87e7f7e141e83689c405b8487e570c0a507","source":{"kind":"arxiv","id":"1905.13648","version":2},"attestation_state":"computed","paper":{"title":"Scene Text Visual Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ali Furkan Biten, Andres Mafla, C.V. Jawahar, Dimosthenis Karatzas, Ernest Valveny, Lluis Gomez, Mar\\c{c}al Rusi\\~nol, Ruben Tito","submitted_at":"2019-05-31T14:47:55Z","abstract_excerpt":"Current visual question answering datasets do not consider the rich semantic information conveyed by text within an image. In this work, we present a new dataset, ST-VQA, that aims to highlight the importance of exploiting high-level semantic information present in images as textual cues in the VQA process. We use this dataset to define a series of tasks of increasing difficulty for which reading the scene text in the context provided by the visual information is necessary to reason and generate an appropriate answer. We propose a new evaluation metric for these tasks to account both for reaso"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1905.13648","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2019-05-31T14:47:55Z","cross_cats_sorted":[],"title_canon_sha256":"c7b619d2370b305db96183abce08d6a54675cdb05ce5bbda6c17bb7cf3aa2236","abstract_canon_sha256":"b8aa0602897ded6f231ef4908262f45e5daeb778d976b881939d0f14e96313a1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:12:26.477250Z","signature_b64":"IJGdxUAU3v57+NJJ2/0DQr05jgP9GkhgT9JgfAUCLwOLPzVXdFF6YHVV9aRqqwvWqjHYKJOaXyaTQp8O/kr2Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"72ca36f343151ce1e05dc818e65de87e7f7e141e83689c405b8487e570c0a507","last_reissued_at":"2026-07-05T00:12:26.476804Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:12:26.476804Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scene Text Visual Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ali Furkan Biten, Andres Mafla, C.V. Jawahar, Dimosthenis Karatzas, Ernest Valveny, Lluis Gomez, Mar\\c{c}al Rusi\\~nol, Ruben Tito","submitted_at":"2019-05-31T14:47:55Z","abstract_excerpt":"Current visual question answering datasets do not consider the rich semantic information conveyed by text within an image. In this work, we present a new dataset, ST-VQA, that aims to highlight the importance of exploiting high-level semantic information present in images as textual cues in the VQA process. We use this dataset to define a series of tasks of increasing difficulty for which reading the scene text in the context provided by the visual information is necessary to reason and generate an appropriate answer. We propose a new evaluation metric for these tasks to account both for reaso"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1905.13648","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1905.13648/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1905.13648","created_at":"2026-07-05T00:12:26.476855+00:00"},{"alias_kind":"arxiv_version","alias_value":"1905.13648v2","created_at":"2026-07-05T00:12:26.476855+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1905.13648","created_at":"2026-07-05T00:12:26.476855+00:00"},{"alias_kind":"pith_short_12","alias_value":"OLFDN42DCUOO","created_at":"2026-07-05T00:12:26.476855+00:00"},{"alias_kind":"pith_short_16","alias_value":"OLFDN42DCUOODYC5","created_at":"2026-07-05T00:12:26.476855+00:00"},{"alias_kind":"pith_short_8","alias_value":"OLFDN42D","created_at":"2026-07-05T00:12:26.476855+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"1907.00490","citing_title":"ICDAR 2019 Competition on Scene Text Visual Question Answering","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2504.09925","citing_title":"FLARE: Fully Integration of Vision-Language Representations for Deep Cross-Modal Understanding","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OLFDN42DCUOODYC5ZAMOMXPIPZ","json":"https://pith.science/pith/OLFDN42DCUOODYC5ZAMOMXPIPZ.json","graph_json":"https://pith.science/api/pith-number/OLFDN42DCUOODYC5ZAMOMXPIPZ/graph.json","events_json":"https://pith.science/api/pith-number/OLFDN42DCUOODYC5ZAMOMXPIPZ/events.json","paper":"https://pith.science/paper/OLFDN42D"},"agent_actions":{"view_html":"https://pith.science/pith/OLFDN42DCUOODYC5ZAMOMXPIPZ","download_json":"https://pith.science/pith/OLFDN42DCUOODYC5ZAMOMXPIPZ.json","view_paper":"https://pith.science/paper/OLFDN42D","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1905.13648&json=true","fetch_graph":"https://pith.science/api/pith-number/OLFDN42DCUOODYC5ZAMOMXPIPZ/graph.json","fetch_events":"https://pith.science/api/pith-number/OLFDN42DCUOODYC5ZAMOMXPIPZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OLFDN42DCUOODYC5ZAMOMXPIPZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OLFDN42DCUOODYC5ZAMOMXPIPZ/action/storage_attestation","attest_author":"https://pith.science/pith/OLFDN42DCUOODYC5ZAMOMXPIPZ/action/author_attestation","sign_citation":"https://pith.science/pith/OLFDN42DCUOODYC5ZAMOMXPIPZ/action/citation_signature","submit_replication":"https://pith.science/pith/OLFDN42DCUOODYC5ZAMOMXPIPZ/action/replication_record"}},"created_at":"2026-07-05T00:12:26.476855+00:00","updated_at":"2026-07-05T00:12:26.476855+00:00"}