{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:YNAISRI23OUQJWJ23HSNYNC27D","short_pith_number":"pith:YNAISRI2","canonical_record":{"source":{"id":"2302.07459","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-02-15T04:25:40Z","cross_cats_sorted":[],"title_canon_sha256":"93429455384dd7cb0e6b2ee4016d13a64a787a29d6f00505fbf96b84a00bd588","abstract_canon_sha256":"be1feaef5ba4f40c94ce530f415d6616ca2a03ce634d80c374e03449047eef31"},"schema_version":"1.0"},"canonical_sha256":"c34089451adba904d93ad9e4dc345af8f2e136989da26f16edb6babea9ea37d2","source":{"kind":"arxiv","id":"2302.07459","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2302.07459","created_at":"2026-07-05T05:43:20Z"},{"alias_kind":"arxiv_version","alias_value":"2302.07459v2","created_at":"2026-07-05T05:43:20Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.07459","created_at":"2026-07-05T05:43:20Z"},{"alias_kind":"pith_short_12","alias_value":"YNAISRI23OUQ","created_at":"2026-07-05T05:43:20Z"},{"alias_kind":"pith_short_16","alias_value":"YNAISRI23OUQJWJ2","created_at":"2026-07-05T05:43:20Z"},{"alias_kind":"pith_short_8","alias_value":"YNAISRI2","created_at":"2026-07-05T05:43:20Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:YNAISRI23OUQJWJ23HSNYNC27D","target":"record","payload":{"canonical_record":{"source":{"id":"2302.07459","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-02-15T04:25:40Z","cross_cats_sorted":[],"title_canon_sha256":"93429455384dd7cb0e6b2ee4016d13a64a787a29d6f00505fbf96b84a00bd588","abstract_canon_sha256":"be1feaef5ba4f40c94ce530f415d6616ca2a03ce634d80c374e03449047eef31"},"schema_version":"1.0"},"canonical_sha256":"c34089451adba904d93ad9e4dc345af8f2e136989da26f16edb6babea9ea37d2","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:43:20.595117Z","signature_b64":"M9ie26FRFiKMJiLft4z3Z2VMqLeaGB0kIELrgBOB3fKp6d+m0ShXthu4Wg58q3BSTNtFG2+yA1KcYRVBoXDRDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c34089451adba904d93ad9e4dc345af8f2e136989da26f16edb6babea9ea37d2","last_reissued_at":"2026-07-05T05:43:20.594601Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:43:20.594601Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2302.07459","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:43:20Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"snVROLrw9Txg9ekv0OT0oI4Uv6yCsbsuXc2eh+Zse/hsAqyHoHwMiIpru6i7RdbPDP4V3vDG3RH8P6h1nNm8Aw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-27T01:03:49.323233Z"},"content_sha256":"08d66b9da1b14d5e054da84543812785b92fbc9a6d207754a767398e84d3da60","schema_version":"1.0","event_id":"sha256:08d66b9da1b14d5e054da84543812785b92fbc9a6d207754a767398e84d3da60"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:YNAISRI23OUQJWJ23HSNYNC27D","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"The Capacity for Moral Self-Correction in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Amanda Askell, Anna Chen, Anna Goldie, Azalia Mirhoseini, Ben Mann, Catherine Olsson, Christopher Olah, Danny Hernandez, Dario Amodei, Dawn Drain, Deep Ganguli, Dustin Li, Eli Tran-Johnson, Ethan Perez, Jack Clark, Jackson Kernion, Jamie Kerr, Jared Kaplan, Jared Mueller, Joshua Landau, Kamal Ndousse, Kamil\\.e Luko\\v{s}i\\=ut\\.e, Karina Nguyen, Liane Lovitt, Michael Sellitto, Nelson Elhage, Nicholas Joseph, Nicholas Schiefer, Noemi Mercado, Nova DasSarma, Oliver Rausch, Robert Lasenby, Robin Larson, Sam McCandlish, Sam Ringer, Samuel R. Bowman, Sandipan Kundu, Saurav Kadavath, Scott Johnston, Shauna Kravec, Sheer El Showk, Tamera Lanham, Thomas I. Liao, Timothy Telleen-Lawton, Tom Brown, Tom Henighan, Tristan Hume, Yuntao Bai, Zac Hatfield-Dodds","submitted_at":"2023-02-15T04:25:40Z","abstract_excerpt":"We test the hypothesis that language models trained with reinforcement learning from human feedback (RLHF) have the capability to \"morally self-correct\" -- to avoid producing harmful outputs -- if instructed to do so. We find strong evidence in support of this hypothesis across three different experiments, each of which reveal different facets of moral self-correction. We find that the capability for moral self-correction emerges at 22B model parameters, and typically improves with increasing model size and RLHF training. We believe that at this level of scale, language models obtain two capab"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.07459","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.07459/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:43:20Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"JNUFHQnSEOZLXtJ1OwdV3jVOtQrPz2retTCnRaKSkRKd27L6uqx5ZgPY88YswAannpyZbl9tHGm2Qzjp7ZzeDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-27T01:03:49.323626Z"},"content_sha256":"3836ec0ad25542ff8782590a739037f681c0522177243d48a957c378af92fa4b","schema_version":"1.0","event_id":"sha256:3836ec0ad25542ff8782590a739037f681c0522177243d48a957c378af92fa4b"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/YNAISRI23OUQJWJ23HSNYNC27D/bundle.json","state_url":"https://pith.science/pith/YNAISRI23OUQJWJ23HSNYNC27D/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/YNAISRI23OUQJWJ23HSNYNC27D/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-07-27T01:03:49Z","links":{"resolver":"https://pith.science/pith/YNAISRI23OUQJWJ23HSNYNC27D","bundle":"https://pith.science/pith/YNAISRI23OUQJWJ23HSNYNC27D/bundle.json","state":"https://pith.science/pith/YNAISRI23OUQJWJ23HSNYNC27D/state.json","well_known_bundle":"https://pith.science/.well-known/pith/YNAISRI23OUQJWJ23HSNYNC27D/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:YNAISRI23OUQJWJ23HSNYNC27D","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"be1feaef5ba4f40c94ce530f415d6616ca2a03ce634d80c374e03449047eef31","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-02-15T04:25:40Z","title_canon_sha256":"93429455384dd7cb0e6b2ee4016d13a64a787a29d6f00505fbf96b84a00bd588"},"schema_version":"1.0","source":{"id":"2302.07459","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2302.07459","created_at":"2026-07-05T05:43:20Z"},{"alias_kind":"arxiv_version","alias_value":"2302.07459v2","created_at":"2026-07-05T05:43:20Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.07459","created_at":"2026-07-05T05:43:20Z"},{"alias_kind":"pith_short_12","alias_value":"YNAISRI23OUQ","created_at":"2026-07-05T05:43:20Z"},{"alias_kind":"pith_short_16","alias_value":"YNAISRI23OUQJWJ2","created_at":"2026-07-05T05:43:20Z"},{"alias_kind":"pith_short_8","alias_value":"YNAISRI2","created_at":"2026-07-05T05:43:20Z"}],"graph_snapshots":[{"event_id":"sha256:3836ec0ad25542ff8782590a739037f681c0522177243d48a957c378af92fa4b","target":"graph","created_at":"2026-07-05T05:43:20Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2302.07459/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"We test the hypothesis that language models trained with reinforcement learning from human feedback (RLHF) have the capability to \"morally self-correct\" -- to avoid producing harmful outputs -- if instructed to do so. We find strong evidence in support of this hypothesis across three different experiments, each of which reveal different facets of moral self-correction. We find that the capability for moral self-correction emerges at 22B model parameters, and typically improves with increasing model size and RLHF training. We believe that at this level of scale, language models obtain two capab","authors_text":"Amanda Askell, Anna Chen, Anna Goldie, Azalia Mirhoseini, Ben Mann, Catherine Olsson, Christopher Olah, Danny Hernandez, Dario Amodei, Dawn Drain, Deep Ganguli, Dustin Li, Eli Tran-Johnson, Ethan Perez, Jack Clark, Jackson Kernion, Jamie Kerr, Jared Kaplan, Jared Mueller, Joshua Landau, Kamal Ndousse, Kamil\\.e Luko\\v{s}i\\=ut\\.e, Karina Nguyen, Liane Lovitt, Michael Sellitto, Nelson Elhage, Nicholas Joseph, Nicholas Schiefer, Noemi Mercado, Nova DasSarma, Oliver Rausch, Robert Lasenby, Robin Larson, Sam McCandlish, Sam Ringer, Samuel R. Bowman, Sandipan Kundu, Saurav Kadavath, Scott Johnston, Shauna Kravec, Sheer El Showk, Tamera Lanham, Thomas I. Liao, Timothy Telleen-Lawton, Tom Brown, Tom Henighan, Tristan Hume, Yuntao Bai, Zac Hatfield-Dodds","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-02-15T04:25:40Z","title":"The Capacity for Moral Self-Correction in Large Language Models"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.07459","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:08d66b9da1b14d5e054da84543812785b92fbc9a6d207754a767398e84d3da60","target":"record","created_at":"2026-07-05T05:43:20Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"be1feaef5ba4f40c94ce530f415d6616ca2a03ce634d80c374e03449047eef31","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-02-15T04:25:40Z","title_canon_sha256":"93429455384dd7cb0e6b2ee4016d13a64a787a29d6f00505fbf96b84a00bd588"},"schema_version":"1.0","source":{"id":"2302.07459","kind":"arxiv","version":2}},"canonical_sha256":"c34089451adba904d93ad9e4dc345af8f2e136989da26f16edb6babea9ea37d2","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"c34089451adba904d93ad9e4dc345af8f2e136989da26f16edb6babea9ea37d2","first_computed_at":"2026-07-05T05:43:20.594601Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T05:43:20.594601Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"M9ie26FRFiKMJiLft4z3Z2VMqLeaGB0kIELrgBOB3fKp6d+m0ShXthu4Wg58q3BSTNtFG2+yA1KcYRVBoXDRDg==","signature_status":"signed_v1","signed_at":"2026-07-05T05:43:20.595117Z","signed_message":"canonical_sha256_bytes"},"source_id":"2302.07459","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:08d66b9da1b14d5e054da84543812785b92fbc9a6d207754a767398e84d3da60","sha256:3836ec0ad25542ff8782590a739037f681c0522177243d48a957c378af92fa4b"],"state_sha256":"dc29534d36ebcc81695bf272c4a7777981c1871e61b0d8edcd6a9aeedb6d4032"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"iqOOBTbSDGUFokaJ9xt6gTZsx4je13aBqJPyyu9nsamRmuc6KfIBIZHgYdDfq+Wyut0HrcArOJvOAI53Ut8zDw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-07-27T01:03:49.325795Z","bundle_sha256":"f8bf0c8953faa2f9fcb1f91c511be59ff88de34341d3a16af89934f43f930a2f"}}