{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:KJTE3IQAIYZBE27I2QOVDVMPV6","short_pith_number":"pith:KJTE3IQA","schema_version":"1.0","canonical_sha256":"52664da2004632126be8d41d51d58faf8bbbee588e08621aba52fadf60f70b03","source":{"kind":"arxiv","id":"2108.01250","version":3},"attestation_state":"computed","paper":{"title":"Your fairness may vary: Pretrained language model fairness in toxic text classification","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Dennis Wei, Ioana Baldini, Karthikeyan Natesan Ramamurthy, Mikhail Yurochkin, Moninder Singh","submitted_at":"2021-08-03T02:16:12Z","abstract_excerpt":"The popularity of pretrained language models in natural language processing systems calls for a careful evaluation of such models in down-stream tasks, which have a higher potential for societal impact. The evaluation of such systems usually focuses on accuracy measures. Our findings in this paper call for attention to be paid to fairness measures as well. Through the analysis of more than a dozen pretrained language models of varying sizes on two toxic text classification tasks (English), we demonstrate that focusing on accuracy measures alone can lead to models with wide variation in fairnes"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2108.01250","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-08-03T02:16:12Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"133afd3761f81b6cde8e549afea99b7c0455931787b71332e469a4cfcc9d58f2","abstract_canon_sha256":"3efcc34560c335a83892b7c850af134042ae7079a66ae60dc8072257ae21fae4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:14:38.297466Z","signature_b64":"Fzi1RWi6up+4vBHMYlno0oTYqxJ4+nbstm51rmbu9nJAlt2FoboPOUSUMPUPSIjCVBDSO/lDfppyqqKiJwMhAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"52664da2004632126be8d41d51d58faf8bbbee588e08621aba52fadf60f70b03","last_reissued_at":"2026-07-05T04:14:38.296978Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:14:38.296978Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Your fairness may vary: Pretrained language model fairness in toxic text classification","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Dennis Wei, Ioana Baldini, Karthikeyan Natesan Ramamurthy, Mikhail Yurochkin, Moninder Singh","submitted_at":"2021-08-03T02:16:12Z","abstract_excerpt":"The popularity of pretrained language models in natural language processing systems calls for a careful evaluation of such models in down-stream tasks, which have a higher potential for societal impact. The evaluation of such systems usually focuses on accuracy measures. Our findings in this paper call for attention to be paid to fairness measures as well. Through the analysis of more than a dozen pretrained language models of varying sizes on two toxic text classification tasks (English), we demonstrate that focusing on accuracy measures alone can lead to models with wide variation in fairnes"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2108.01250","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2108.01250/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2108.01250","created_at":"2026-07-05T04:14:38.297034+00:00"},{"alias_kind":"arxiv_version","alias_value":"2108.01250v3","created_at":"2026-07-05T04:14:38.297034+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2108.01250","created_at":"2026-07-05T04:14:38.297034+00:00"},{"alias_kind":"pith_short_12","alias_value":"KJTE3IQAIYZB","created_at":"2026-07-05T04:14:38.297034+00:00"},{"alias_kind":"pith_short_16","alias_value":"KJTE3IQAIYZBE27I","created_at":"2026-07-05T04:14:38.297034+00:00"},{"alias_kind":"pith_short_8","alias_value":"KJTE3IQA","created_at":"2026-07-05T04:14:38.297034+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.05610","citing_title":"Mitigating Confounding in Speech-Based Dementia Detection through Weight Masking","ref_index":1,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KJTE3IQAIYZBE27I2QOVDVMPV6","json":"https://pith.science/pith/KJTE3IQAIYZBE27I2QOVDVMPV6.json","graph_json":"https://pith.science/api/pith-number/KJTE3IQAIYZBE27I2QOVDVMPV6/graph.json","events_json":"https://pith.science/api/pith-number/KJTE3IQAIYZBE27I2QOVDVMPV6/events.json","paper":"https://pith.science/paper/KJTE3IQA"},"agent_actions":{"view_html":"https://pith.science/pith/KJTE3IQAIYZBE27I2QOVDVMPV6","download_json":"https://pith.science/pith/KJTE3IQAIYZBE27I2QOVDVMPV6.json","view_paper":"https://pith.science/paper/KJTE3IQA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2108.01250&json=true","fetch_graph":"https://pith.science/api/pith-number/KJTE3IQAIYZBE27I2QOVDVMPV6/graph.json","fetch_events":"https://pith.science/api/pith-number/KJTE3IQAIYZBE27I2QOVDVMPV6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KJTE3IQAIYZBE27I2QOVDVMPV6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KJTE3IQAIYZBE27I2QOVDVMPV6/action/storage_attestation","attest_author":"https://pith.science/pith/KJTE3IQAIYZBE27I2QOVDVMPV6/action/author_attestation","sign_citation":"https://pith.science/pith/KJTE3IQAIYZBE27I2QOVDVMPV6/action/citation_signature","submit_replication":"https://pith.science/pith/KJTE3IQAIYZBE27I2QOVDVMPV6/action/replication_record"}},"created_at":"2026-07-05T04:14:38.297034+00:00","updated_at":"2026-07-05T04:14:38.297034+00:00"}