{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:WRAFQY5PEAZTDUCICUEOSOUGF7","short_pith_number":"pith:WRAFQY5P","schema_version":"1.0","canonical_sha256":"b4405863af203331d0481508e93a862fcaa05cb8a641f24509933bbde00c005f","source":{"kind":"arxiv","id":"2503.02157","version":1},"attestation_state":"computed","paper":{"title":"MedHEval: Benchmarking Hallucinations and Mitigation Strategies in Medical Large Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Aofei Chang, Cao Xiao, Fenglong Ma, Le Huang, Parminder Bhatia, Taha Kass-hout","submitted_at":"2025-03-04T00:40:09Z","abstract_excerpt":"Large Vision Language Models (LVLMs) are becoming increasingly important in the medical domain, yet Medical LVLMs (Med-LVLMs) frequently generate hallucinations due to limited expertise and the complexity of medical applications. Existing benchmarks fail to effectively evaluate hallucinations based on their underlying causes and lack assessments of mitigation strategies. To address this gap, we introduce MedHEval, a novel benchmark that systematically evaluates hallucinations and mitigation strategies in Med-LVLMs by categorizing them into three underlying causes: visual misinterpretation, kno"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.02157","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-04T00:40:09Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"d0e3705095fc7a680703b234b568db447be030a7f7a21b9cb37a10c6228adbbb","abstract_canon_sha256":"baef373c921e5599b9cce6bcf98640ef4b0b417a3ab1fb7d13fc6b4fdcd7fd7b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:23:37.983224Z","signature_b64":"ItzgyTU7Cm5Vl7vufc7i0w8PTFO2xKES7H0g5XWBxNStQx3RlSuj8uHbUb+NeLWfq+9qz+ZCt9osVkgwHLYMAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b4405863af203331d0481508e93a862fcaa05cb8a641f24509933bbde00c005f","last_reissued_at":"2026-07-05T10:23:37.982224Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:23:37.982224Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MedHEval: Benchmarking Hallucinations and Mitigation Strategies in Medical Large Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Aofei Chang, Cao Xiao, Fenglong Ma, Le Huang, Parminder Bhatia, Taha Kass-hout","submitted_at":"2025-03-04T00:40:09Z","abstract_excerpt":"Large Vision Language Models (LVLMs) are becoming increasingly important in the medical domain, yet Medical LVLMs (Med-LVLMs) frequently generate hallucinations due to limited expertise and the complexity of medical applications. Existing benchmarks fail to effectively evaluate hallucinations based on their underlying causes and lack assessments of mitigation strategies. To address this gap, we introduce MedHEval, a novel benchmark that systematically evaluates hallucinations and mitigation strategies in Med-LVLMs by categorizing them into three underlying causes: visual misinterpretation, kno"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.02157","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.02157/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.02157","created_at":"2026-07-05T10:23:37.982352+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.02157v1","created_at":"2026-07-05T10:23:37.982352+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.02157","created_at":"2026-07-05T10:23:37.982352+00:00"},{"alias_kind":"pith_short_12","alias_value":"WRAFQY5PEAZT","created_at":"2026-07-05T10:23:37.982352+00:00"},{"alias_kind":"pith_short_16","alias_value":"WRAFQY5PEAZTDUCI","created_at":"2026-07-05T10:23:37.982352+00:00"},{"alias_kind":"pith_short_8","alias_value":"WRAFQY5P","created_at":"2026-07-05T10:23:37.982352+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08399","citing_title":"Prompt Compression via Activation Aggregation","ref_index":78,"is_internal_anchor":true},{"citing_arxiv_id":"2607.00060","citing_title":"Synergistic Perception-Reasoning Governance: Grounding Medical MLLMs with Verifiable Anatomical Evidence","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2510.02001","citing_title":"Generating Findings for Jaw Cysts in Dental Panoramic Radiographs Using a GPT-Based VLM: A Preliminary Study on Building a Two-Stage Self-Correction Loop with Structured Output (SLSO) Framework","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WRAFQY5PEAZTDUCICUEOSOUGF7","json":"https://pith.science/pith/WRAFQY5PEAZTDUCICUEOSOUGF7.json","graph_json":"https://pith.science/api/pith-number/WRAFQY5PEAZTDUCICUEOSOUGF7/graph.json","events_json":"https://pith.science/api/pith-number/WRAFQY5PEAZTDUCICUEOSOUGF7/events.json","paper":"https://pith.science/paper/WRAFQY5P"},"agent_actions":{"view_html":"https://pith.science/pith/WRAFQY5PEAZTDUCICUEOSOUGF7","download_json":"https://pith.science/pith/WRAFQY5PEAZTDUCICUEOSOUGF7.json","view_paper":"https://pith.science/paper/WRAFQY5P","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.02157&json=true","fetch_graph":"https://pith.science/api/pith-number/WRAFQY5PEAZTDUCICUEOSOUGF7/graph.json","fetch_events":"https://pith.science/api/pith-number/WRAFQY5PEAZTDUCICUEOSOUGF7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WRAFQY5PEAZTDUCICUEOSOUGF7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WRAFQY5PEAZTDUCICUEOSOUGF7/action/storage_attestation","attest_author":"https://pith.science/pith/WRAFQY5PEAZTDUCICUEOSOUGF7/action/author_attestation","sign_citation":"https://pith.science/pith/WRAFQY5PEAZTDUCICUEOSOUGF7/action/citation_signature","submit_replication":"https://pith.science/pith/WRAFQY5PEAZTDUCICUEOSOUGF7/action/replication_record"}},"created_at":"2026-07-05T10:23:37.982352+00:00","updated_at":"2026-07-05T10:23:37.982352+00:00"}