{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BVNHNHO64P5T5OJKLMG2IIZ2LS","short_pith_number":"pith:BVNHNHO6","schema_version":"1.0","canonical_sha256":"0d5a769ddee3fb3eb92a5b0da4233a5c928ca1864c0e0a3b83f719b6f0ea92ee","source":{"kind":"arxiv","id":"2505.22298","version":1},"attestation_state":"computed","paper":{"title":"Adaptive Detoxification: Safeguarding General Capabilities of LLMs through Toxicity-Aware Knowledge Editing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Fangming Liu, Jing Li, Jun Yu, Meishan Zhang, Min Zhang, Wenya Wang, Xiucheng Li, Yifan Lu, Yigeng Zhou, Yihui Zhang","submitted_at":"2025-05-28T12:37:06Z","abstract_excerpt":"Large language models (LLMs) exhibit impressive language capabilities but remain vulnerable to malicious prompts and jailbreaking attacks. Existing knowledge editing methods for LLM detoxification face two major challenges. First, they often rely on entity-specific localization, making them ineffective against adversarial inputs without explicit entities. Second, these methods suffer from over-editing, where detoxified models reject legitimate queries, compromising overall performance. In this paper, we propose ToxEdit, a toxicity-aware knowledge editing approach that dynamically detects toxic"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.22298","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-28T12:37:06Z","cross_cats_sorted":[],"title_canon_sha256":"dd309997f89bf45188755ebb7d03e0028624ce86f8fbcd398ed901e9e65e7384","abstract_canon_sha256":"7a47e8c3d4f54348729199f2b82c64aa8525b6038989489a0a6e6c5b1d40d92b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:11:12.829924Z","signature_b64":"T5BUKu3pDcL3gxzwEBBXSorJOG8XOgpnmrWfvm0twIcWyI+mFt47v0lM316e9U5IpIRFVU2BYCLOA/+109N5Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0d5a769ddee3fb3eb92a5b0da4233a5c928ca1864c0e0a3b83f719b6f0ea92ee","last_reissued_at":"2026-07-05T11:11:12.829404Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:11:12.829404Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Adaptive Detoxification: Safeguarding General Capabilities of LLMs through Toxicity-Aware Knowledge Editing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Fangming Liu, Jing Li, Jun Yu, Meishan Zhang, Min Zhang, Wenya Wang, Xiucheng Li, Yifan Lu, Yigeng Zhou, Yihui Zhang","submitted_at":"2025-05-28T12:37:06Z","abstract_excerpt":"Large language models (LLMs) exhibit impressive language capabilities but remain vulnerable to malicious prompts and jailbreaking attacks. Existing knowledge editing methods for LLM detoxification face two major challenges. First, they often rely on entity-specific localization, making them ineffective against adversarial inputs without explicit entities. Second, these methods suffer from over-editing, where detoxified models reject legitimate queries, compromising overall performance. In this paper, we propose ToxEdit, a toxicity-aware knowledge editing approach that dynamically detects toxic"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.22298","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.22298/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.22298","created_at":"2026-07-05T11:11:12.829470+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.22298v1","created_at":"2026-07-05T11:11:12.829470+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.22298","created_at":"2026-07-05T11:11:12.829470+00:00"},{"alias_kind":"pith_short_12","alias_value":"BVNHNHO64P5T","created_at":"2026-07-05T11:11:12.829470+00:00"},{"alias_kind":"pith_short_16","alias_value":"BVNHNHO64P5T5OJK","created_at":"2026-07-05T11:11:12.829470+00:00"},{"alias_kind":"pith_short_8","alias_value":"BVNHNHO6","created_at":"2026-07-05T11:11:12.829470+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.27997","citing_title":"Where Does Toxicity Live? Mechanistic Localization and Targeted Suppression in Language Models","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BVNHNHO64P5T5OJKLMG2IIZ2LS","json":"https://pith.science/pith/BVNHNHO64P5T5OJKLMG2IIZ2LS.json","graph_json":"https://pith.science/api/pith-number/BVNHNHO64P5T5OJKLMG2IIZ2LS/graph.json","events_json":"https://pith.science/api/pith-number/BVNHNHO64P5T5OJKLMG2IIZ2LS/events.json","paper":"https://pith.science/paper/BVNHNHO6"},"agent_actions":{"view_html":"https://pith.science/pith/BVNHNHO64P5T5OJKLMG2IIZ2LS","download_json":"https://pith.science/pith/BVNHNHO64P5T5OJKLMG2IIZ2LS.json","view_paper":"https://pith.science/paper/BVNHNHO6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.22298&json=true","fetch_graph":"https://pith.science/api/pith-number/BVNHNHO64P5T5OJKLMG2IIZ2LS/graph.json","fetch_events":"https://pith.science/api/pith-number/BVNHNHO64P5T5OJKLMG2IIZ2LS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BVNHNHO64P5T5OJKLMG2IIZ2LS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BVNHNHO64P5T5OJKLMG2IIZ2LS/action/storage_attestation","attest_author":"https://pith.science/pith/BVNHNHO64P5T5OJKLMG2IIZ2LS/action/author_attestation","sign_citation":"https://pith.science/pith/BVNHNHO64P5T5OJKLMG2IIZ2LS/action/citation_signature","submit_replication":"https://pith.science/pith/BVNHNHO64P5T5OJKLMG2IIZ2LS/action/replication_record"}},"created_at":"2026-07-05T11:11:12.829470+00:00","updated_at":"2026-07-05T11:11:12.829470+00:00"}