{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:MTFOVMS57FGQYBKSW3TEI4BNXD","short_pith_number":"pith:MTFOVMS5","schema_version":"1.0","canonical_sha256":"64caeab25df94d0c0552b6e644702db8f41e3a1ee1e2b781c79d5dbf4c057a6c","source":{"kind":"arxiv","id":"2310.06552","version":3},"attestation_state":"computed","paper":{"title":"Automated clinical coding using off-the-shelf large language models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Alison Q. O'Neil, Antanas Kascenas, Joseph S. Boyle, Maria Liakata, Pat Lok","submitted_at":"2023-10-10T11:56:48Z","abstract_excerpt":"The task of assigning diagnostic ICD codes to patient hospital admissions is typically performed by expert human coders. Efforts towards automated ICD coding are dominated by supervised deep learning models. However, difficulties in learning to predict the large number of rare codes remain a barrier to adoption in clinical practice. In this work, we leverage off-the-shelf pre-trained generative large language models (LLMs) to develop a practical solution that is suitable for zero-shot and few-shot code assignment, with no need for further task-specific training. Unsupervised pre-training alone"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.06552","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2023-10-10T11:56:48Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"7c0992dab23505476c8bc0b7e87f32e65762ee1479dcc97555ab638c3e6c4abe","abstract_canon_sha256":"9c597787fadf1bb085c609aeff2208bcd811760b1bd32e7013df5b380b6ed115"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:12:00.288005Z","signature_b64":"dhsW83cOcf/YUgSlXjC3WwRVhXOsIQCSeJpt4rVU54mAQHV4qWa8y76HVQUf+Au5sFfYTjrqoLIE0rg0VMv4DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"64caeab25df94d0c0552b6e644702db8f41e3a1ee1e2b781c79d5dbf4c057a6c","last_reissued_at":"2026-07-05T07:12:00.287643Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:12:00.287643Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Automated clinical coding using off-the-shelf large language models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Alison Q. O'Neil, Antanas Kascenas, Joseph S. Boyle, Maria Liakata, Pat Lok","submitted_at":"2023-10-10T11:56:48Z","abstract_excerpt":"The task of assigning diagnostic ICD codes to patient hospital admissions is typically performed by expert human coders. Efforts towards automated ICD coding are dominated by supervised deep learning models. However, difficulties in learning to predict the large number of rare codes remain a barrier to adoption in clinical practice. In this work, we leverage off-the-shelf pre-trained generative large language models (LLMs) to develop a practical solution that is suitable for zero-shot and few-shot code assignment, with no need for further task-specific training. Unsupervised pre-training alone"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.06552","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.06552/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.06552","created_at":"2026-07-05T07:12:00.287708+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.06552v3","created_at":"2026-07-05T07:12:00.287708+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.06552","created_at":"2026-07-05T07:12:00.287708+00:00"},{"alias_kind":"pith_short_12","alias_value":"MTFOVMS57FGQ","created_at":"2026-07-05T07:12:00.287708+00:00"},{"alias_kind":"pith_short_16","alias_value":"MTFOVMS57FGQYBKS","created_at":"2026-07-05T07:12:00.287708+00:00"},{"alias_kind":"pith_short_8","alias_value":"MTFOVMS5","created_at":"2026-07-05T07:12:00.287708+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.22768","citing_title":"Secure On-Premise Deployment of Open-Weights Large Language Models in Radiology: An Isolation-First Architecture with Prospective Pilot Evaluation","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MTFOVMS57FGQYBKSW3TEI4BNXD","json":"https://pith.science/pith/MTFOVMS57FGQYBKSW3TEI4BNXD.json","graph_json":"https://pith.science/api/pith-number/MTFOVMS57FGQYBKSW3TEI4BNXD/graph.json","events_json":"https://pith.science/api/pith-number/MTFOVMS57FGQYBKSW3TEI4BNXD/events.json","paper":"https://pith.science/paper/MTFOVMS5"},"agent_actions":{"view_html":"https://pith.science/pith/MTFOVMS57FGQYBKSW3TEI4BNXD","download_json":"https://pith.science/pith/MTFOVMS57FGQYBKSW3TEI4BNXD.json","view_paper":"https://pith.science/paper/MTFOVMS5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.06552&json=true","fetch_graph":"https://pith.science/api/pith-number/MTFOVMS57FGQYBKSW3TEI4BNXD/graph.json","fetch_events":"https://pith.science/api/pith-number/MTFOVMS57FGQYBKSW3TEI4BNXD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MTFOVMS57FGQYBKSW3TEI4BNXD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MTFOVMS57FGQYBKSW3TEI4BNXD/action/storage_attestation","attest_author":"https://pith.science/pith/MTFOVMS57FGQYBKSW3TEI4BNXD/action/author_attestation","sign_citation":"https://pith.science/pith/MTFOVMS57FGQYBKSW3TEI4BNXD/action/citation_signature","submit_replication":"https://pith.science/pith/MTFOVMS57FGQYBKSW3TEI4BNXD/action/replication_record"}},"created_at":"2026-07-05T07:12:00.287708+00:00","updated_at":"2026-07-05T07:12:00.287708+00:00"}