{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:MLAKXYO3HC4RYJFZ7GF6VGDNNZ","short_pith_number":"pith:MLAKXYO3","schema_version":"1.0","canonical_sha256":"62c0abe1db38b91c24b9f98bea986d6e6d624f47e95cbe7661a93b964dea44cf","source":{"kind":"arxiv","id":"2311.16452","version":1},"attestation_state":"computed","paper":{"title":"Can Generalist Foundation Models Outcompete Special-Purpose Tuning? Case Study in Medicine","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chris White, Dean Carignan, Eric Horvitz, Harsha Nori, Hoifung Poon, Jonathan Larson, Naoto Usuyama, Nicholas King, Nicolo Fusi, Renqian Luo, Richard Edgar, Robert Osazuwa Ness, Scott Mayer McKinney, Sheng Zhang, Tao Qin, Weishung Liu, Yin Tat Lee, Yuanzhi Li","submitted_at":"2023-11-28T03:16:12Z","abstract_excerpt":"Generalist foundation models such as GPT-4 have displayed surprising capabilities in a wide variety of domains and tasks. Yet, there is a prevalent assumption that they cannot match specialist capabilities of fine-tuned models. For example, most explorations to date on medical competency benchmarks have leveraged domain-specific training, as exemplified by efforts on BioGPT and Med-PaLM. We build on a prior study of GPT-4's capabilities on medical challenge benchmarks in the absence of special training. Rather than using simple prompting to highlight the model's out-of-the-box capabilities, we"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.16452","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-11-28T03:16:12Z","cross_cats_sorted":[],"title_canon_sha256":"c43f47bb5c1efefa513a9db8c154cefacf0570a60d27dd871c0887f477adf065","abstract_canon_sha256":"fe4f03c0244f93270b75761aa6664ba926bc87dc0571ea9b8dc48b4d4c9792e0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:17:41.918704Z","signature_b64":"V4sfTPQiJhNubPmkQNnxGz7qqV+cYhGfxYTn/Bx/RmQa+85fnYhMM6O3mFOGnw8oFOu1g/0PxyDDY3/ppCV0DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"62c0abe1db38b91c24b9f98bea986d6e6d624f47e95cbe7661a93b964dea44cf","last_reissued_at":"2026-07-05T07:17:41.918145Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:17:41.918145Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can Generalist Foundation Models Outcompete Special-Purpose Tuning? Case Study in Medicine","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chris White, Dean Carignan, Eric Horvitz, Harsha Nori, Hoifung Poon, Jonathan Larson, Naoto Usuyama, Nicholas King, Nicolo Fusi, Renqian Luo, Richard Edgar, Robert Osazuwa Ness, Scott Mayer McKinney, Sheng Zhang, Tao Qin, Weishung Liu, Yin Tat Lee, Yuanzhi Li","submitted_at":"2023-11-28T03:16:12Z","abstract_excerpt":"Generalist foundation models such as GPT-4 have displayed surprising capabilities in a wide variety of domains and tasks. Yet, there is a prevalent assumption that they cannot match specialist capabilities of fine-tuned models. For example, most explorations to date on medical competency benchmarks have leveraged domain-specific training, as exemplified by efforts on BioGPT and Med-PaLM. We build on a prior study of GPT-4's capabilities on medical challenge benchmarks in the absence of special training. Rather than using simple prompting to highlight the model's out-of-the-box capabilities, we"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.16452","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.16452/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.16452","created_at":"2026-07-05T07:17:41.918208+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.16452v1","created_at":"2026-07-05T07:17:41.918208+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.16452","created_at":"2026-07-05T07:17:41.918208+00:00"},{"alias_kind":"pith_short_12","alias_value":"MLAKXYO3HC4R","created_at":"2026-07-05T07:17:41.918208+00:00"},{"alias_kind":"pith_short_16","alias_value":"MLAKXYO3HC4RYJFZ","created_at":"2026-07-05T07:17:41.918208+00:00"},{"alias_kind":"pith_short_8","alias_value":"MLAKXYO3","created_at":"2026-07-05T07:17:41.918208+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":33,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06157","citing_title":"LLM Agents for Deliberative Collaboration: A Study on Joint Decision Making Under Partial Observability","ref_index":74,"is_internal_anchor":true},{"citing_arxiv_id":"2607.01440","citing_title":"FaithMed: Training LLMs For Faithful Evidence-Based Medical Reasoning","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12854","citing_title":"Small LLMs for Biomedical Claim Verification: Cost-Effective Fine-Tuning, Structural Dataset Shortcuts, and Cross-Domain Generalization","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05241","citing_title":"Search-Time Contamination in Deep Research Agents: Measuring Performance Inflation in Public Benchmark Evaluation","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05481","citing_title":"Towards Unified and Data-Efficient Prognostics and Health Management with Tabular Foundation Models","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31410","citing_title":"FAM-Bench: A Multimodal Benchmark for Condition-Aware Food-as-Medicine Reasoning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28332","citing_title":"When Medical Safety Alignment Fails: A Benchmark for Evaluating LLMs on High-Risk Medical Queries","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28900","citing_title":"MedEvoEval: Evaluating Continual Evolution of Doctor Agents through Simulated Clinical Episodes","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25878","citing_title":"A Clinically Validated Foundation Model for Comprehensive Lung Pathology Interpretation","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29368","citing_title":"SURGENT: A Surgical Multi-Agent Assistance System Across the Perioperative Workflow","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2401.07345","citing_title":"Can an LLM Learn Preferences from Choice Data?","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2410.21276","citing_title":"GPT-4o System Card","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2502.07143","citing_title":"Ask Patients with Patience: Enabling LLMs for Human-Centric Medical Dialogue with Grounded Reasoning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2504.01990","citing_title":"Advances and Challenges in Foundation Agents: From Brain-Inspired Intelligence to Evolutionary, Collaborative, and Safe Systems","ref_index":183,"is_internal_anchor":false},{"citing_arxiv_id":"2405.07960","citing_title":"AgentClinic: a multimodal agent benchmark to evaluate AI in simulated clinical environments","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2603.05308","citing_title":"Med-V1: Small Language Models for Zero-shot and Scalable Biomedical Evidence Attribution","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20425","citing_title":"AgentCo-op: Retrieval-Based Synthesis of Interoperable Multi-Agent Workflows","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20525","citing_title":"NeuroQA: A Large-Scale Image-Grounded Benchmark for 3D Brain MRI Understanding","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16630","citing_title":"PrivScope: Task-scoped Disclosure Control for Hybrid Agentic Systems","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2508.05012","citing_title":"Making Prompts First-Class Citizens for Adaptive LLM Pipelines","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2509.24186","citing_title":"Measuring Competency, Not Performance: Item-Aware Evaluation Across Medical Benchmarks","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2404.18416","citing_title":"Capabilities of Gemini Models in Medicine","ref_index":179,"is_internal_anchor":false},{"citing_arxiv_id":"2412.18925","citing_title":"HuatuoGPT-o1, Towards Medical Complex Reasoning with LLMs","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08559","citing_title":"Medical Reasoning with Large Language Models: A Survey and MR-Bench","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2406.07496","citing_title":"TextGrad: Automatic \"Differentiation\" via Text","ref_index":66,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MLAKXYO3HC4RYJFZ7GF6VGDNNZ","json":"https://pith.science/pith/MLAKXYO3HC4RYJFZ7GF6VGDNNZ.json","graph_json":"https://pith.science/api/pith-number/MLAKXYO3HC4RYJFZ7GF6VGDNNZ/graph.json","events_json":"https://pith.science/api/pith-number/MLAKXYO3HC4RYJFZ7GF6VGDNNZ/events.json","paper":"https://pith.science/paper/MLAKXYO3"},"agent_actions":{"view_html":"https://pith.science/pith/MLAKXYO3HC4RYJFZ7GF6VGDNNZ","download_json":"https://pith.science/pith/MLAKXYO3HC4RYJFZ7GF6VGDNNZ.json","view_paper":"https://pith.science/paper/MLAKXYO3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.16452&json=true","fetch_graph":"https://pith.science/api/pith-number/MLAKXYO3HC4RYJFZ7GF6VGDNNZ/graph.json","fetch_events":"https://pith.science/api/pith-number/MLAKXYO3HC4RYJFZ7GF6VGDNNZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MLAKXYO3HC4RYJFZ7GF6VGDNNZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MLAKXYO3HC4RYJFZ7GF6VGDNNZ/action/storage_attestation","attest_author":"https://pith.science/pith/MLAKXYO3HC4RYJFZ7GF6VGDNNZ/action/author_attestation","sign_citation":"https://pith.science/pith/MLAKXYO3HC4RYJFZ7GF6VGDNNZ/action/citation_signature","submit_replication":"https://pith.science/pith/MLAKXYO3HC4RYJFZ7GF6VGDNNZ/action/replication_record"}},"created_at":"2026-07-05T07:17:41.918208+00:00","updated_at":"2026-07-05T07:17:41.918208+00:00"}