{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LD7FO4TLP6JPZQPVOOT7M6BEZX","short_pith_number":"pith:LD7FO4TL","schema_version":"1.0","canonical_sha256":"58fe57726b7f92fcc1f573a7f67824cdc316dd73c0c8c4179b7ea9b89b73ff09","source":{"kind":"arxiv","id":"2401.12208","version":2},"attestation_state":"computed","paper":{"title":"A Vision-Language Foundation Model to Enhance Efficiency of Chest X-ray Interpretation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Akshay S. Chaudhari, Alaa Youssef, Andrew Johnston, Cameron Olsen, Christian Bluethgen, Christopher F. Beaulieu, Curtis P. Langlotz, Dave Van Veen, Eduardo Pontes Reis, Emily B. Tsai, Jean-Benoit Delbrouck, Jenia Jitsev, Jeya Maria Jose Valanarasu, Joseph Paul Cohen, Justin Xu, Louis Blankemeier, Magdalini Paschali, Maya Varma, Mohamed Siddig Eltayeb Muneer, Sergios Gatidis, Stephan Altmayer, Tanishq Mathew Abraham, Zhihong Chen","submitted_at":"2024-01-22T18:51:07Z","abstract_excerpt":"Over 1.4 billion chest X-rays (CXRs) are performed annually due to their cost-effectiveness as an initial diagnostic test. This scale of radiological studies provides a significant opportunity to streamline CXR interpretation and documentation. While foundation models are a promising solution, the lack of publicly available large-scale datasets and benchmarks inhibits their iterative development and real-world evaluation. To overcome these challenges, we constructed a large-scale dataset (CheXinstruct), which we utilized to train a vision-language foundation model (CheXagent). We systematicall"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.12208","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-01-22T18:51:07Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"86d1f4c7ce048651f2955bc46aa79a105d65e4ae25665dbe7b4726e68ee84f03","abstract_canon_sha256":"8a442969971dd3fda036e402ca81c796f508f0bfd15263ae5213cac3612f031a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:51:30.488201Z","signature_b64":"ZtgZ9OI+2BhQKzZFauaRJHp6BEmYG17RgX8FvWuscNPit97QB8rEz46Z/4yiG8ch7gYF16Ir3zfBkLV2ATWwCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"58fe57726b7f92fcc1f573a7f67824cdc316dd73c0c8c4179b7ea9b89b73ff09","last_reissued_at":"2026-07-05T09:51:30.487763Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:51:30.487763Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Vision-Language Foundation Model to Enhance Efficiency of Chest X-ray Interpretation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Akshay S. Chaudhari, Alaa Youssef, Andrew Johnston, Cameron Olsen, Christian Bluethgen, Christopher F. Beaulieu, Curtis P. Langlotz, Dave Van Veen, Eduardo Pontes Reis, Emily B. Tsai, Jean-Benoit Delbrouck, Jenia Jitsev, Jeya Maria Jose Valanarasu, Joseph Paul Cohen, Justin Xu, Louis Blankemeier, Magdalini Paschali, Maya Varma, Mohamed Siddig Eltayeb Muneer, Sergios Gatidis, Stephan Altmayer, Tanishq Mathew Abraham, Zhihong Chen","submitted_at":"2024-01-22T18:51:07Z","abstract_excerpt":"Over 1.4 billion chest X-rays (CXRs) are performed annually due to their cost-effectiveness as an initial diagnostic test. This scale of radiological studies provides a significant opportunity to streamline CXR interpretation and documentation. While foundation models are a promising solution, the lack of publicly available large-scale datasets and benchmarks inhibits their iterative development and real-world evaluation. To overcome these challenges, we constructed a large-scale dataset (CheXinstruct), which we utilized to train a vision-language foundation model (CheXagent). We systematicall"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.12208","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.12208/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.12208","created_at":"2026-07-05T09:51:30.487825+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.12208v2","created_at":"2026-07-05T09:51:30.487825+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.12208","created_at":"2026-07-05T09:51:30.487825+00:00"},{"alias_kind":"pith_short_12","alias_value":"LD7FO4TLP6JP","created_at":"2026-07-05T09:51:30.487825+00:00"},{"alias_kind":"pith_short_16","alias_value":"LD7FO4TLP6JPZQPV","created_at":"2026-07-05T09:51:30.487825+00:00"},{"alias_kind":"pith_short_8","alias_value":"LD7FO4TL","created_at":"2026-07-05T09:51:30.487825+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":24,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05880","citing_title":"Harrison.Rad 1.5 Technical Report: A radiology foundation model that can draft reports from images, priors and clinical context","ref_index":1,"is_internal_anchor":true},{"citing_arxiv_id":"2606.21290","citing_title":"NoduLoCC2026: Lung Nodule Localization and Classification Contest from Chest X-Ray Images","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21020","citing_title":"CheXpercept: A Benchmark for Evaluating Expert-Level Lesion Perception in Chest X-rays","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12590","citing_title":"Analyzing and Improving Fine-grained Preference Optimization in Medical LVLMs","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06407","citing_title":"A Vision-language Framework for Comparative Reasoning in Radiology","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26691","citing_title":"Mind the Tool Failures: Achieving Synergistic Tool Gains for Medical Agents","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23629","citing_title":"DDX-TRACE: A Benchmark for Medical Diagnostic Trajectories in VLMs","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2401.02458","citing_title":"Data-Centric Foundation Models in Computational Healthcare: A Survey","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2408.16213","citing_title":"M4CXR: Exploring Multi-task Potentials of Multi-modal Large Language Models for Chest X-ray Interpretation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2504.07415","citing_title":"RA-RRG: Multimodal Retrieval-Augmented Radiology Report Generation with Key Phrase Extraction","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09450","citing_title":"ECHO: Efficient Chest X-ray Report Generation with One-step Block Diffusion","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20469","citing_title":"HalluCXR: Benchmarking and Mitigating Hallucinations in Medical Vision-Language Models for Chest Radiograph Interpretation","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2506.09067","citing_title":"Enhancing the Safety of Medical Vision-Language Models by Synthetic Demonstrations","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2509.20490","citing_title":"RadAgents: Multimodal Agentic Reasoning for Chest X-ray Interpretation with Radiologist-like Workflows","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2510.07191","citing_title":"Resolution scaling governs DINOv3 transfer performance in chest radiograph classification","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2511.15825","citing_title":"IMACT-CXR: An Interactive Multi-Agent Conversational Tutoring System for Chest X-Ray Interpretation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2305.10415","citing_title":"PMC-VQA: Visual Instruction Tuning for Medical Visual Question Answering","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13060","citing_title":"Dental-TriageBench: Benchmarking Multimodal Reasoning for Hierarchical Dental Triage","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22989","citing_title":"CheXmix: Unified Generative Pretraining for Vision Language Models in Medical Imaging","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05810","citing_title":"CXR-ContraBench: Benchmarking Negated-Option Attraction in Medical VLMs","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09450","citing_title":"ECHO: Efficient Chest X-ray Report Generation with One-step Block Diffusion","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13598","citing_title":"Enhancing Reinforcement Learning for Radiology Report Generation with Evidence-aware Rewards and Self-correcting Preference Learning","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18250","citing_title":"Medical Image Understanding Improves Survival Prediction via Visual Instruction Tuning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18967","citing_title":"CXRMate-2: Structured Multimodal Temporal Embeddings and Tractable Reinforcement Learning for Clinically Acceptable Chest X-ray Radiology Report Generation","ref_index":65,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LD7FO4TLP6JPZQPVOOT7M6BEZX","json":"https://pith.science/pith/LD7FO4TLP6JPZQPVOOT7M6BEZX.json","graph_json":"https://pith.science/api/pith-number/LD7FO4TLP6JPZQPVOOT7M6BEZX/graph.json","events_json":"https://pith.science/api/pith-number/LD7FO4TLP6JPZQPVOOT7M6BEZX/events.json","paper":"https://pith.science/paper/LD7FO4TL"},"agent_actions":{"view_html":"https://pith.science/pith/LD7FO4TLP6JPZQPVOOT7M6BEZX","download_json":"https://pith.science/pith/LD7FO4TLP6JPZQPVOOT7M6BEZX.json","view_paper":"https://pith.science/paper/LD7FO4TL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.12208&json=true","fetch_graph":"https://pith.science/api/pith-number/LD7FO4TLP6JPZQPVOOT7M6BEZX/graph.json","fetch_events":"https://pith.science/api/pith-number/LD7FO4TLP6JPZQPVOOT7M6BEZX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LD7FO4TLP6JPZQPVOOT7M6BEZX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LD7FO4TLP6JPZQPVOOT7M6BEZX/action/storage_attestation","attest_author":"https://pith.science/pith/LD7FO4TLP6JPZQPVOOT7M6BEZX/action/author_attestation","sign_citation":"https://pith.science/pith/LD7FO4TLP6JPZQPVOOT7M6BEZX/action/citation_signature","submit_replication":"https://pith.science/pith/LD7FO4TLP6JPZQPVOOT7M6BEZX/action/replication_record"}},"created_at":"2026-07-05T09:51:30.487825+00:00","updated_at":"2026-07-05T09:51:30.487825+00:00"}