{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:HCDK5IX45PCYNEJH4PC666V5MW","short_pith_number":"pith:HCDK5IX4","schema_version":"1.0","canonical_sha256":"3886aea2fcebc5869127e3c5ef7abd65b05d9a9f3b735cc7bab4da9de6888ec6","source":{"kind":"arxiv","id":"2306.00526","version":4},"attestation_state":"computed","paper":{"title":"Layout and Task Aware Instruction Prompt for Zero-shot Document Image Question Answering","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CL","authors_text":"Wenjin Wang, Yin Zhang, Yixin Ou, Yunhao Li","submitted_at":"2023-06-01T10:28:12Z","abstract_excerpt":"Layout-aware pre-trained models has achieved significant progress on document image question answering. They introduce extra learnable modules into existing language models to capture layout information within document images from text bounding box coordinates obtained by OCR tools. However, extra modules necessitate pre-training on extensive document images. This prevents these methods from directly utilizing off-the-shelf instruction-tuning language foundation models, which have recently shown promising potential in zero-shot learning. Instead, in this paper, we find that instruction-tuning "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.00526","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2023-06-01T10:28:12Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"84258574dade786bffef0e6bb7d79fe15ef804838751d73d08a5e01c9e3cbe76","abstract_canon_sha256":"14274c3d8a7c1eb976e8a43114a3bfc49cc5dcbf7e71b5217b183bc758b28b04"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:48:35.031058Z","signature_b64":"PpKNUkAXodF9ucnhMoAZGze1NvrDfRflCdF+2eIpAkhmUqCIkOvNVPHoEpYuaVfMSmPzw0UY+y7qXyweOsEvAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3886aea2fcebc5869127e3c5ef7abd65b05d9a9f3b735cc7bab4da9de6888ec6","last_reissued_at":"2026-07-05T06:48:35.030588Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:48:35.030588Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Layout and Task Aware Instruction Prompt for Zero-shot Document Image Question Answering","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CL","authors_text":"Wenjin Wang, Yin Zhang, Yixin Ou, Yunhao Li","submitted_at":"2023-06-01T10:28:12Z","abstract_excerpt":"Layout-aware pre-trained models has achieved significant progress on document image question answering. They introduce extra learnable modules into existing language models to capture layout information within document images from text bounding box coordinates obtained by OCR tools. However, extra modules necessitate pre-training on extensive document images. This prevents these methods from directly utilizing off-the-shelf instruction-tuning language foundation models, which have recently shown promising potential in zero-shot learning. Instead, in this paper, we find that instruction-tuning "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.00526","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.00526/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.00526","created_at":"2026-07-05T06:48:35.030642+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.00526v4","created_at":"2026-07-05T06:48:35.030642+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.00526","created_at":"2026-07-05T06:48:35.030642+00:00"},{"alias_kind":"pith_short_12","alias_value":"HCDK5IX45PCY","created_at":"2026-07-05T06:48:35.030642+00:00"},{"alias_kind":"pith_short_16","alias_value":"HCDK5IX45PCYNEJH","created_at":"2026-07-05T06:48:35.030642+00:00"},{"alias_kind":"pith_short_8","alias_value":"HCDK5IX4","created_at":"2026-07-05T06:48:35.030642+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.18603","citing_title":"Doc-CoB: Enhancing Document Understanding with Visual Chain-of-Boxes Reasoning","ref_index":49,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HCDK5IX45PCYNEJH4PC666V5MW","json":"https://pith.science/pith/HCDK5IX45PCYNEJH4PC666V5MW.json","graph_json":"https://pith.science/api/pith-number/HCDK5IX45PCYNEJH4PC666V5MW/graph.json","events_json":"https://pith.science/api/pith-number/HCDK5IX45PCYNEJH4PC666V5MW/events.json","paper":"https://pith.science/paper/HCDK5IX4"},"agent_actions":{"view_html":"https://pith.science/pith/HCDK5IX45PCYNEJH4PC666V5MW","download_json":"https://pith.science/pith/HCDK5IX45PCYNEJH4PC666V5MW.json","view_paper":"https://pith.science/paper/HCDK5IX4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.00526&json=true","fetch_graph":"https://pith.science/api/pith-number/HCDK5IX45PCYNEJH4PC666V5MW/graph.json","fetch_events":"https://pith.science/api/pith-number/HCDK5IX45PCYNEJH4PC666V5MW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HCDK5IX45PCYNEJH4PC666V5MW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HCDK5IX45PCYNEJH4PC666V5MW/action/storage_attestation","attest_author":"https://pith.science/pith/HCDK5IX45PCYNEJH4PC666V5MW/action/author_attestation","sign_citation":"https://pith.science/pith/HCDK5IX45PCYNEJH4PC666V5MW/action/citation_signature","submit_replication":"https://pith.science/pith/HCDK5IX45PCYNEJH4PC666V5MW/action/replication_record"}},"created_at":"2026-07-05T06:48:35.030642+00:00","updated_at":"2026-07-05T06:48:35.030642+00:00"}