{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:6F64HNGFGAWN67PK2BLO3445VS","short_pith_number":"pith:6F64HNGF","schema_version":"1.0","canonical_sha256":"f17dc3b4c5302cdf7dead056edf39dacab5aa2d35cbef4e3d941cb730b3853bb","source":{"kind":"arxiv","id":"2301.08986","version":1},"attestation_state":"computed","paper":{"title":"Adapting a Language Model While Preserving its General Knowledge","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.NE"],"primary_cat":"cs.CL","authors_text":"Bing Liu, Haowei Lin, Hu Xu, Lei Shu, Yijia Shao, Zixuan Ke","submitted_at":"2023-01-21T17:57:53Z","abstract_excerpt":"Domain-adaptive pre-training (or DA-training for short), also known as post-training, aims to train a pre-trained general-purpose language model (LM) using an unlabeled corpus of a particular domain to adapt the LM so that end-tasks in the domain can give improved performances. However, existing DA-training methods are in some sense blind as they do not explicitly identify what knowledge in the LM should be preserved and what should be changed by the domain corpus. This paper shows that the existing methods are suboptimal and proposes a novel method to perform a more informed adaptation of the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2301.08986","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.CL","submitted_at":"2023-01-21T17:57:53Z","cross_cats_sorted":["cs.AI","cs.LG","cs.NE"],"title_canon_sha256":"5a5ed5915bc4dac652e3414e08d18d8bb12da32514132e7c80b717ede655e51d","abstract_canon_sha256":"ea89ed998af5ebe01933779e67847f9721f2c92183d2b617320521568958a71e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:34:54.908285Z","signature_b64":"j9ap3bGeV5QJccMOmIdA+0N6y+OyafEqTPWkjMYEFr8eFs5JfoyaX3eosc94liOzFZ80zh/NHPB8nR5ZgQJlBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f17dc3b4c5302cdf7dead056edf39dacab5aa2d35cbef4e3d941cb730b3853bb","last_reissued_at":"2026-07-05T05:34:54.907742Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:34:54.907742Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Adapting a Language Model While Preserving its General Knowledge","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.NE"],"primary_cat":"cs.CL","authors_text":"Bing Liu, Haowei Lin, Hu Xu, Lei Shu, Yijia Shao, Zixuan Ke","submitted_at":"2023-01-21T17:57:53Z","abstract_excerpt":"Domain-adaptive pre-training (or DA-training for short), also known as post-training, aims to train a pre-trained general-purpose language model (LM) using an unlabeled corpus of a particular domain to adapt the LM so that end-tasks in the domain can give improved performances. However, existing DA-training methods are in some sense blind as they do not explicitly identify what knowledge in the LM should be preserved and what should be changed by the domain corpus. This paper shows that the existing methods are suboptimal and proposes a novel method to perform a more informed adaptation of the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2301.08986","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2301.08986/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2301.08986","created_at":"2026-07-05T05:34:54.907817+00:00"},{"alias_kind":"arxiv_version","alias_value":"2301.08986v1","created_at":"2026-07-05T05:34:54.907817+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2301.08986","created_at":"2026-07-05T05:34:54.907817+00:00"},{"alias_kind":"pith_short_12","alias_value":"6F64HNGFGAWN","created_at":"2026-07-05T05:34:54.907817+00:00"},{"alias_kind":"pith_short_16","alias_value":"6F64HNGFGAWN67PK","created_at":"2026-07-05T05:34:54.907817+00:00"},{"alias_kind":"pith_short_8","alias_value":"6F64HNGF","created_at":"2026-07-05T05:34:54.907817+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.17840","citing_title":"Learning Beyond the Surface: How Far Can Continual Pre-Training with LoRA Enhance LLMs' Domain-Specific Insight Learning?","ref_index":11,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6F64HNGFGAWN67PK2BLO3445VS","json":"https://pith.science/pith/6F64HNGFGAWN67PK2BLO3445VS.json","graph_json":"https://pith.science/api/pith-number/6F64HNGFGAWN67PK2BLO3445VS/graph.json","events_json":"https://pith.science/api/pith-number/6F64HNGFGAWN67PK2BLO3445VS/events.json","paper":"https://pith.science/paper/6F64HNGF"},"agent_actions":{"view_html":"https://pith.science/pith/6F64HNGFGAWN67PK2BLO3445VS","download_json":"https://pith.science/pith/6F64HNGFGAWN67PK2BLO3445VS.json","view_paper":"https://pith.science/paper/6F64HNGF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2301.08986&json=true","fetch_graph":"https://pith.science/api/pith-number/6F64HNGFGAWN67PK2BLO3445VS/graph.json","fetch_events":"https://pith.science/api/pith-number/6F64HNGFGAWN67PK2BLO3445VS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6F64HNGFGAWN67PK2BLO3445VS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6F64HNGFGAWN67PK2BLO3445VS/action/storage_attestation","attest_author":"https://pith.science/pith/6F64HNGFGAWN67PK2BLO3445VS/action/author_attestation","sign_citation":"https://pith.science/pith/6F64HNGFGAWN67PK2BLO3445VS/action/citation_signature","submit_replication":"https://pith.science/pith/6F64HNGFGAWN67PK2BLO3445VS/action/replication_record"}},"created_at":"2026-07-05T05:34:54.907817+00:00","updated_at":"2026-07-05T05:34:54.907817+00:00"}