{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:VHSKLCPN5CAQZ3B7TY27KJUIXS","short_pith_number":"pith:VHSKLCPN","schema_version":"1.0","canonical_sha256":"a9e4a589ede8810cec3f9e35f52688bca17be8937a6ff96165957a3334e9c0d3","source":{"kind":"arxiv","id":"2207.11850","version":1},"attestation_state":"computed","paper":{"title":"Visual Perturbation-aware Collaborative Learning for Overcoming the Language Prior Problem","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jianhua Yin, Jianlong Wu, Liqiang Nie, Yan Yan, Yudong Han","submitted_at":"2022-07-24T23:50:52Z","abstract_excerpt":"Several studies have recently pointed that existing Visual Question Answering (VQA) models heavily suffer from the language prior problem, which refers to capturing superficial statistical correlations between the question type and the answer whereas ignoring the image contents. Numerous efforts have been dedicated to strengthen the image dependency by creating the delicate models or introducing the extra visual annotations. However, these methods cannot sufficiently explore how the visual cues explicitly affect the learned answer representation, which is vital for language reliance alleviatio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2207.11850","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-07-24T23:50:52Z","cross_cats_sorted":[],"title_canon_sha256":"6b4d08988b62313f7630ee49f4dcda3053156a2f9dcf5c5e7d559c6bfc2cadf7","abstract_canon_sha256":"a8c021559b91ea7c0947f84289494b6bc104da28553b4431175ba0895ab51c28"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:43:10.286167Z","signature_b64":"hGAKkYZIpa227cqk2Yz2vdz5V+cdB3W0VBv8IXD60+W9/l+XLLOzQyxRqAJwA6IPEXDnCN8v/Vq9Snnrss7vCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a9e4a589ede8810cec3f9e35f52688bca17be8937a6ff96165957a3334e9c0d3","last_reissued_at":"2026-07-05T04:43:10.285780Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:43:10.285780Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Visual Perturbation-aware Collaborative Learning for Overcoming the Language Prior Problem","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jianhua Yin, Jianlong Wu, Liqiang Nie, Yan Yan, Yudong Han","submitted_at":"2022-07-24T23:50:52Z","abstract_excerpt":"Several studies have recently pointed that existing Visual Question Answering (VQA) models heavily suffer from the language prior problem, which refers to capturing superficial statistical correlations between the question type and the answer whereas ignoring the image contents. Numerous efforts have been dedicated to strengthen the image dependency by creating the delicate models or introducing the extra visual annotations. However, these methods cannot sufficiently explore how the visual cues explicitly affect the learned answer representation, which is vital for language reliance alleviatio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2207.11850","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2207.11850/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2207.11850","created_at":"2026-07-05T04:43:10.285838+00:00"},{"alias_kind":"arxiv_version","alias_value":"2207.11850v1","created_at":"2026-07-05T04:43:10.285838+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2207.11850","created_at":"2026-07-05T04:43:10.285838+00:00"},{"alias_kind":"pith_short_12","alias_value":"VHSKLCPN5CAQ","created_at":"2026-07-05T04:43:10.285838+00:00"},{"alias_kind":"pith_short_16","alias_value":"VHSKLCPN5CAQZ3B7","created_at":"2026-07-05T04:43:10.285838+00:00"},{"alias_kind":"pith_short_8","alias_value":"VHSKLCPN","created_at":"2026-07-05T04:43:10.285838+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28401","citing_title":"Vision-driven Preference Synthesis for Mitigating Hallucinations in VLMs","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2507.12455","citing_title":"Mitigating Object Hallucinations via Sentence-Level Early Intervention","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21300","citing_title":"Reducing Object Hallucination in LVLMs via Emphasizing Image-negative Tokens","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VHSKLCPN5CAQZ3B7TY27KJUIXS","json":"https://pith.science/pith/VHSKLCPN5CAQZ3B7TY27KJUIXS.json","graph_json":"https://pith.science/api/pith-number/VHSKLCPN5CAQZ3B7TY27KJUIXS/graph.json","events_json":"https://pith.science/api/pith-number/VHSKLCPN5CAQZ3B7TY27KJUIXS/events.json","paper":"https://pith.science/paper/VHSKLCPN"},"agent_actions":{"view_html":"https://pith.science/pith/VHSKLCPN5CAQZ3B7TY27KJUIXS","download_json":"https://pith.science/pith/VHSKLCPN5CAQZ3B7TY27KJUIXS.json","view_paper":"https://pith.science/paper/VHSKLCPN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2207.11850&json=true","fetch_graph":"https://pith.science/api/pith-number/VHSKLCPN5CAQZ3B7TY27KJUIXS/graph.json","fetch_events":"https://pith.science/api/pith-number/VHSKLCPN5CAQZ3B7TY27KJUIXS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VHSKLCPN5CAQZ3B7TY27KJUIXS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VHSKLCPN5CAQZ3B7TY27KJUIXS/action/storage_attestation","attest_author":"https://pith.science/pith/VHSKLCPN5CAQZ3B7TY27KJUIXS/action/author_attestation","sign_citation":"https://pith.science/pith/VHSKLCPN5CAQZ3B7TY27KJUIXS/action/citation_signature","submit_replication":"https://pith.science/pith/VHSKLCPN5CAQZ3B7TY27KJUIXS/action/replication_record"}},"created_at":"2026-07-05T04:43:10.285838+00:00","updated_at":"2026-07-05T04:43:10.285838+00:00"}