{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:Y5V4RG67JIBFMFL3DRT2JSPYNG","short_pith_number":"pith:Y5V4RG67","schema_version":"1.0","canonical_sha256":"c76bc89bdf4a0256157b1c67a4c9f8699bca73e074a06be63410ceadf7eea81b","source":{"kind":"arxiv","id":"2002.02497","version":2},"attestation_state":"computed","paper":{"title":"On the limits of cross-domain generalization in automated X-ray prediction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","q-bio.QM","stat.ML"],"primary_cat":"eess.IV","authors_text":"Hadrien Bertrand, Joseph Paul Cohen, Mohammad Hashir, Rupert Brooks","submitted_at":"2020-02-06T20:07:54Z","abstract_excerpt":"This large scale study focuses on quantifying what X-rays diagnostic prediction tasks generalize well across multiple different datasets. We present evidence that the issue of generalization is not due to a shift in the images but instead a shift in the labels. We study the cross-domain performance, agreement between models, and model representations. We find interesting discrepancies between performance and agreement where models which both achieve good performance disagree in their predictions as well as models which agree yet achieve poor performance. We also test for concept similarity by "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2002.02497","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.IV","submitted_at":"2020-02-06T20:07:54Z","cross_cats_sorted":["cs.LG","q-bio.QM","stat.ML"],"title_canon_sha256":"ecb11b16bcd8619d707f6df13cb170a76f306922b65cf639c7f0f6f2a38436db","abstract_canon_sha256":"05b3c24b41c88558987ae8b95067676ff29f2b1e1447a5ae1679f7dcd2772577"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:05:26.007020Z","signature_b64":"dmZs7KjJU9NKMbFDHzkWs8gxDDs222KMLfuDepZiAxvIEkNk2tiU8n3ou8ZuQeoBpG9loy3NW1sKC85MgiZjBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c76bc89bdf4a0256157b1c67a4c9f8699bca73e074a06be63410ceadf7eea81b","last_reissued_at":"2026-07-05T01:05:26.006535Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:05:26.006535Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the limits of cross-domain generalization in automated X-ray prediction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","q-bio.QM","stat.ML"],"primary_cat":"eess.IV","authors_text":"Hadrien Bertrand, Joseph Paul Cohen, Mohammad Hashir, Rupert Brooks","submitted_at":"2020-02-06T20:07:54Z","abstract_excerpt":"This large scale study focuses on quantifying what X-rays diagnostic prediction tasks generalize well across multiple different datasets. We present evidence that the issue of generalization is not due to a shift in the images but instead a shift in the labels. We study the cross-domain performance, agreement between models, and model representations. We find interesting discrepancies between performance and agreement where models which both achieve good performance disagree in their predictions as well as models which agree yet achieve poor performance. We also test for concept similarity by "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2002.02497","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2002.02497/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2002.02497","created_at":"2026-07-05T01:05:26.006593+00:00"},{"alias_kind":"arxiv_version","alias_value":"2002.02497v2","created_at":"2026-07-05T01:05:26.006593+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2002.02497","created_at":"2026-07-05T01:05:26.006593+00:00"},{"alias_kind":"pith_short_12","alias_value":"Y5V4RG67JIBF","created_at":"2026-07-05T01:05:26.006593+00:00"},{"alias_kind":"pith_short_16","alias_value":"Y5V4RG67JIBFMFL3","created_at":"2026-07-05T01:05:26.006593+00:00"},{"alias_kind":"pith_short_8","alias_value":"Y5V4RG67","created_at":"2026-07-05T01:05:26.006593+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03211","citing_title":"Optimized Labeling Resource Allocation for Prediction-Assisted Inference via OPAL","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31785","citing_title":"Self-Supervised Temporal Regularization for Landmark-Based Cardiac Segmentation with Automatic AHA Regional Mapping","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04614","citing_title":"A Clinical Point Cloud Paradigm for In-Hospital Mortality Prediction from Multi-Level Incomplete Multimodal EHRs","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Y5V4RG67JIBFMFL3DRT2JSPYNG","json":"https://pith.science/pith/Y5V4RG67JIBFMFL3DRT2JSPYNG.json","graph_json":"https://pith.science/api/pith-number/Y5V4RG67JIBFMFL3DRT2JSPYNG/graph.json","events_json":"https://pith.science/api/pith-number/Y5V4RG67JIBFMFL3DRT2JSPYNG/events.json","paper":"https://pith.science/paper/Y5V4RG67"},"agent_actions":{"view_html":"https://pith.science/pith/Y5V4RG67JIBFMFL3DRT2JSPYNG","download_json":"https://pith.science/pith/Y5V4RG67JIBFMFL3DRT2JSPYNG.json","view_paper":"https://pith.science/paper/Y5V4RG67","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2002.02497&json=true","fetch_graph":"https://pith.science/api/pith-number/Y5V4RG67JIBFMFL3DRT2JSPYNG/graph.json","fetch_events":"https://pith.science/api/pith-number/Y5V4RG67JIBFMFL3DRT2JSPYNG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Y5V4RG67JIBFMFL3DRT2JSPYNG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Y5V4RG67JIBFMFL3DRT2JSPYNG/action/storage_attestation","attest_author":"https://pith.science/pith/Y5V4RG67JIBFMFL3DRT2JSPYNG/action/author_attestation","sign_citation":"https://pith.science/pith/Y5V4RG67JIBFMFL3DRT2JSPYNG/action/citation_signature","submit_replication":"https://pith.science/pith/Y5V4RG67JIBFMFL3DRT2JSPYNG/action/replication_record"}},"created_at":"2026-07-05T01:05:26.006593+00:00","updated_at":"2026-07-05T01:05:26.006593+00:00"}