{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:AVXYEG2LY4IZTVNGSWES7QFF4X","short_pith_number":"pith:AVXYEG2L","schema_version":"1.0","canonical_sha256":"056f821b4bc71199d5a695892fc0a5e5ebabc5cdd75c6f22d46cf1a97248cb9f","source":{"kind":"arxiv","id":"2502.11441","version":3},"attestation_state":"computed","paper":{"title":"Which Retain Set Matters for LLM Unlearning? A Case Study on Entity Unlearning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hwan Chang, Hwanhee Lee","submitted_at":"2025-02-17T04:55:02Z","abstract_excerpt":"Large language models (LLMs) risk retaining unauthorized or sensitive information from their training data, which raises privacy concerns. LLM unlearning seeks to mitigate these risks by selectively removing specified data while maintaining overall model performance. However, most existing work focus on methods to achieve effective forgetting and does not provide a detailed analysis of the retain set, the portion of training data that is not targeted for removal. In this paper, we investigate the effects of unlearning on various subsets of the retain set through a case study on entity unlearni"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.11441","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-17T04:55:02Z","cross_cats_sorted":[],"title_canon_sha256":"81d55d73ab7d380f8859b543187202461b2f14e6ed66d2313caaeac9c59d1106","abstract_canon_sha256":"9a4b9345d738c70439ab9edc4ffeef665cc212e84681876e78759fbd7926aa5a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:10:52.630583Z","signature_b64":"HcY/yQNAc+sq7zbi4d2FFiY06HoAJT8dBA1lcpZ6WTdwuSW6kzHq5XOnTMaA6KK2TFeqd7Q68SveM016a1H0Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"056f821b4bc71199d5a695892fc0a5e5ebabc5cdd75c6f22d46cf1a97248cb9f","last_reissued_at":"2026-07-05T11:10:52.630048Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:10:52.630048Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Which Retain Set Matters for LLM Unlearning? A Case Study on Entity Unlearning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hwan Chang, Hwanhee Lee","submitted_at":"2025-02-17T04:55:02Z","abstract_excerpt":"Large language models (LLMs) risk retaining unauthorized or sensitive information from their training data, which raises privacy concerns. LLM unlearning seeks to mitigate these risks by selectively removing specified data while maintaining overall model performance. However, most existing work focus on methods to achieve effective forgetting and does not provide a detailed analysis of the retain set, the portion of training data that is not targeted for removal. In this paper, we investigate the effects of unlearning on various subsets of the retain set through a case study on entity unlearni"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.11441","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.11441/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.11441","created_at":"2026-07-05T11:10:52.630115+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.11441v3","created_at":"2026-07-05T11:10:52.630115+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.11441","created_at":"2026-07-05T11:10:52.630115+00:00"},{"alias_kind":"pith_short_12","alias_value":"AVXYEG2LY4IZ","created_at":"2026-07-05T11:10:52.630115+00:00"},{"alias_kind":"pith_short_16","alias_value":"AVXYEG2LY4IZTVNG","created_at":"2026-07-05T11:10:52.630115+00:00"},{"alias_kind":"pith_short_8","alias_value":"AVXYEG2L","created_at":"2026-07-05T11:10:52.630115+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27379","citing_title":"Position: The Term \"Machine Unlearning\" Is Overused in LLMs","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AVXYEG2LY4IZTVNGSWES7QFF4X","json":"https://pith.science/pith/AVXYEG2LY4IZTVNGSWES7QFF4X.json","graph_json":"https://pith.science/api/pith-number/AVXYEG2LY4IZTVNGSWES7QFF4X/graph.json","events_json":"https://pith.science/api/pith-number/AVXYEG2LY4IZTVNGSWES7QFF4X/events.json","paper":"https://pith.science/paper/AVXYEG2L"},"agent_actions":{"view_html":"https://pith.science/pith/AVXYEG2LY4IZTVNGSWES7QFF4X","download_json":"https://pith.science/pith/AVXYEG2LY4IZTVNGSWES7QFF4X.json","view_paper":"https://pith.science/paper/AVXYEG2L","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.11441&json=true","fetch_graph":"https://pith.science/api/pith-number/AVXYEG2LY4IZTVNGSWES7QFF4X/graph.json","fetch_events":"https://pith.science/api/pith-number/AVXYEG2LY4IZTVNGSWES7QFF4X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AVXYEG2LY4IZTVNGSWES7QFF4X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AVXYEG2LY4IZTVNGSWES7QFF4X/action/storage_attestation","attest_author":"https://pith.science/pith/AVXYEG2LY4IZTVNGSWES7QFF4X/action/author_attestation","sign_citation":"https://pith.science/pith/AVXYEG2LY4IZTVNGSWES7QFF4X/action/citation_signature","submit_replication":"https://pith.science/pith/AVXYEG2LY4IZTVNGSWES7QFF4X/action/replication_record"}},"created_at":"2026-07-05T11:10:52.630115+00:00","updated_at":"2026-07-05T11:10:52.630115+00:00"}