{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:OJSQ77L6W4DE7HRUIY55O7HTYD","short_pith_number":"pith:OJSQ77L6","schema_version":"1.0","canonical_sha256":"72650ffd7eb7064f9e34463bd77cf3c0efcaf0762e426d2dd29983bf216fc05f","source":{"kind":"arxiv","id":"2109.07445","version":1},"attestation_state":"computed","paper":{"title":"Challenges in Detoxifying Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CY","cs.LG"],"primary_cat":"cs.CL","authors_text":"Amelia Glaese, Ben Coppin, Johannes Welbl, John Mellor, Jonathan Uesato, Kirsty Anderson, Lisa Anne Hendricks, Po-Sen Huang, Pushmeet Kohli, Sumanth Dathathri","submitted_at":"2021-09-15T17:27:06Z","abstract_excerpt":"Large language models (LM) generate remarkably fluent text and can be efficiently adapted across NLP tasks. Measuring and guaranteeing the quality of generated text in terms of safety is imperative for deploying LMs in the real world; to this end, prior work often relies on automatic evaluation of LM toxicity. We critically discuss this approach, evaluate several toxicity mitigation strategies with respect to both automatic and human evaluation, and analyze consequences of toxicity mitigation in terms of model bias and LM quality. We demonstrate that while basic intervention strategies can eff"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2109.07445","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-09-15T17:27:06Z","cross_cats_sorted":["cs.AI","cs.CY","cs.LG"],"title_canon_sha256":"eae2fa313e6e2974d7c344f0203984c38948bec391c8ee581b836f2ad501e1af","abstract_canon_sha256":"6fed863490e6d9dbc2df1703ff7223a0e2bb31d56d3611219b20300c4989dfc3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:14:49.280388Z","signature_b64":"EZAWYiZbeQQk//pgOsQ+0+8vvukzFpu9NnezWE089GwK/zFV/TN46cX113ZSNOW1pB5SdBvSqtiuTkP78nNEDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"72650ffd7eb7064f9e34463bd77cf3c0efcaf0762e426d2dd29983bf216fc05f","last_reissued_at":"2026-07-05T03:14:49.279898Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:14:49.279898Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Challenges in Detoxifying Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CY","cs.LG"],"primary_cat":"cs.CL","authors_text":"Amelia Glaese, Ben Coppin, Johannes Welbl, John Mellor, Jonathan Uesato, Kirsty Anderson, Lisa Anne Hendricks, Po-Sen Huang, Pushmeet Kohli, Sumanth Dathathri","submitted_at":"2021-09-15T17:27:06Z","abstract_excerpt":"Large language models (LM) generate remarkably fluent text and can be efficiently adapted across NLP tasks. Measuring and guaranteeing the quality of generated text in terms of safety is imperative for deploying LMs in the real world; to this end, prior work often relies on automatic evaluation of LM toxicity. We critically discuss this approach, evaluate several toxicity mitigation strategies with respect to both automatic and human evaluation, and analyze consequences of toxicity mitigation in terms of model bias and LM quality. We demonstrate that while basic intervention strategies can eff"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2109.07445","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2109.07445/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2109.07445","created_at":"2026-07-05T03:14:49.279958+00:00"},{"alias_kind":"arxiv_version","alias_value":"2109.07445v1","created_at":"2026-07-05T03:14:49.279958+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2109.07445","created_at":"2026-07-05T03:14:49.279958+00:00"},{"alias_kind":"pith_short_12","alias_value":"OJSQ77L6W4DE","created_at":"2026-07-05T03:14:49.279958+00:00"},{"alias_kind":"pith_short_16","alias_value":"OJSQ77L6W4DE7HRU","created_at":"2026-07-05T03:14:49.279958+00:00"},{"alias_kind":"pith_short_8","alias_value":"OJSQ77L6","created_at":"2026-07-05T03:14:49.279958+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.27997","citing_title":"Where Does Toxicity Live? Mechanistic Localization and Targeted Suppression in Language Models","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2201.11990","citing_title":"Using DeepSpeed and Megatron to Train Megatron-Turing NLG 530B, A Large-Scale Generative Language Model","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2401.05561","citing_title":"TrustLLM: Trustworthiness in Large Language Models","ref_index":246,"is_internal_anchor":false},{"citing_arxiv_id":"2309.10253","citing_title":"GPTFUZZER: Red Teaming Large Language Models with Auto-Generated Jailbreak Prompts","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2112.04359","citing_title":"Ethical and social risks of harm from Language Models","ref_index":290,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22089","citing_title":"Ethics Testing: Proactive Identification of Generative AI System Harms","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2112.00861","citing_title":"A General Language Assistant as a Laboratory for Alignment","ref_index":252,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OJSQ77L6W4DE7HRUIY55O7HTYD","json":"https://pith.science/pith/OJSQ77L6W4DE7HRUIY55O7HTYD.json","graph_json":"https://pith.science/api/pith-number/OJSQ77L6W4DE7HRUIY55O7HTYD/graph.json","events_json":"https://pith.science/api/pith-number/OJSQ77L6W4DE7HRUIY55O7HTYD/events.json","paper":"https://pith.science/paper/OJSQ77L6"},"agent_actions":{"view_html":"https://pith.science/pith/OJSQ77L6W4DE7HRUIY55O7HTYD","download_json":"https://pith.science/pith/OJSQ77L6W4DE7HRUIY55O7HTYD.json","view_paper":"https://pith.science/paper/OJSQ77L6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2109.07445&json=true","fetch_graph":"https://pith.science/api/pith-number/OJSQ77L6W4DE7HRUIY55O7HTYD/graph.json","fetch_events":"https://pith.science/api/pith-number/OJSQ77L6W4DE7HRUIY55O7HTYD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OJSQ77L6W4DE7HRUIY55O7HTYD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OJSQ77L6W4DE7HRUIY55O7HTYD/action/storage_attestation","attest_author":"https://pith.science/pith/OJSQ77L6W4DE7HRUIY55O7HTYD/action/author_attestation","sign_citation":"https://pith.science/pith/OJSQ77L6W4DE7HRUIY55O7HTYD/action/citation_signature","submit_replication":"https://pith.science/pith/OJSQ77L6W4DE7HRUIY55O7HTYD/action/replication_record"}},"created_at":"2026-07-05T03:14:49.279958+00:00","updated_at":"2026-07-05T03:14:49.279958+00:00"}