{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EKUBAFFS2FIBWPCMVYPTUDWYZ2","short_pith_number":"pith:EKUBAFFS","schema_version":"1.0","canonical_sha256":"22a81014b2d1501b3c4cae1f3a0ed8ceb7ed9decb8b822e01fbf44a0e3870cab","source":{"kind":"arxiv","id":"2411.03700","version":2},"attestation_state":"computed","paper":{"title":"The Root Shapes the Fruit: On the Persistence of Gender-Exclusive Harms in Aligned Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Adina Williams, Anaelia Ovalle, Eric Michael Smith, Kai-Wei Chang, Krunoslav Lehman Pavasovic, Levent Sagun, Louis Martin, Luke Zettlemoyer","submitted_at":"2024-11-06T06:50:50Z","abstract_excerpt":"Natural-language assistants are designed to provide users with helpful responses while avoiding harmful outputs, largely achieved through alignment to human preferences. Yet there is limited understanding of whether alignment techniques may inadvertently perpetuate or even amplify harmful biases inherited from their pre-aligned base models. This issue is compounded by the choice of bias evaluation benchmarks in popular preference-finetuned models, which predominantly focus on dominant social categories, such as binary gender, thereby limiting insights into biases affecting underrepresented gro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.03700","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-11-06T06:50:50Z","cross_cats_sorted":[],"title_canon_sha256":"8b2106a424ce548d813a16d598e5d8655abbf8bb91a31c75f6a2179ee4929545","abstract_canon_sha256":"ea188247a584cd74343da104c0603a10e6af0238cd9b7954d83a7f57be001251"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:01:28.428076Z","signature_b64":"0XgldqJ0CH6bA0yZalwD3lE16iqbAspcy46qG7TnUbxh1LK2+nnqRUWnGsFis9BR5iN6YutRoQOpQuEP9HrBBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"22a81014b2d1501b3c4cae1f3a0ed8ceb7ed9decb8b822e01fbf44a0e3870cab","last_reissued_at":"2026-07-05T11:01:28.427605Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:01:28.427605Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Root Shapes the Fruit: On the Persistence of Gender-Exclusive Harms in Aligned Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Adina Williams, Anaelia Ovalle, Eric Michael Smith, Kai-Wei Chang, Krunoslav Lehman Pavasovic, Levent Sagun, Louis Martin, Luke Zettlemoyer","submitted_at":"2024-11-06T06:50:50Z","abstract_excerpt":"Natural-language assistants are designed to provide users with helpful responses while avoiding harmful outputs, largely achieved through alignment to human preferences. Yet there is limited understanding of whether alignment techniques may inadvertently perpetuate or even amplify harmful biases inherited from their pre-aligned base models. This issue is compounded by the choice of bias evaluation benchmarks in popular preference-finetuned models, which predominantly focus on dominant social categories, such as binary gender, thereby limiting insights into biases affecting underrepresented gro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.03700","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.03700/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.03700","created_at":"2026-07-05T11:01:28.427667+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.03700v2","created_at":"2026-07-05T11:01:28.427667+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.03700","created_at":"2026-07-05T11:01:28.427667+00:00"},{"alias_kind":"pith_short_12","alias_value":"EKUBAFFS2FIB","created_at":"2026-07-05T11:01:28.427667+00:00"},{"alias_kind":"pith_short_16","alias_value":"EKUBAFFS2FIBWPCM","created_at":"2026-07-05T11:01:28.427667+00:00"},{"alias_kind":"pith_short_8","alias_value":"EKUBAFFS","created_at":"2026-07-05T11:01:28.427667+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.14080","citing_title":"Gender Trouble in Language Models: An Empirical Audit Guided by Gender Performativity Theory","ref_index":33,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EKUBAFFS2FIBWPCMVYPTUDWYZ2","json":"https://pith.science/pith/EKUBAFFS2FIBWPCMVYPTUDWYZ2.json","graph_json":"https://pith.science/api/pith-number/EKUBAFFS2FIBWPCMVYPTUDWYZ2/graph.json","events_json":"https://pith.science/api/pith-number/EKUBAFFS2FIBWPCMVYPTUDWYZ2/events.json","paper":"https://pith.science/paper/EKUBAFFS"},"agent_actions":{"view_html":"https://pith.science/pith/EKUBAFFS2FIBWPCMVYPTUDWYZ2","download_json":"https://pith.science/pith/EKUBAFFS2FIBWPCMVYPTUDWYZ2.json","view_paper":"https://pith.science/paper/EKUBAFFS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.03700&json=true","fetch_graph":"https://pith.science/api/pith-number/EKUBAFFS2FIBWPCMVYPTUDWYZ2/graph.json","fetch_events":"https://pith.science/api/pith-number/EKUBAFFS2FIBWPCMVYPTUDWYZ2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EKUBAFFS2FIBWPCMVYPTUDWYZ2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EKUBAFFS2FIBWPCMVYPTUDWYZ2/action/storage_attestation","attest_author":"https://pith.science/pith/EKUBAFFS2FIBWPCMVYPTUDWYZ2/action/author_attestation","sign_citation":"https://pith.science/pith/EKUBAFFS2FIBWPCMVYPTUDWYZ2/action/citation_signature","submit_replication":"https://pith.science/pith/EKUBAFFS2FIBWPCMVYPTUDWYZ2/action/replication_record"}},"created_at":"2026-07-05T11:01:28.427667+00:00","updated_at":"2026-07-05T11:01:28.427667+00:00"}