{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:DQIAHD7TFHTSSSKBLO5BZFSRGI","short_pith_number":"pith:DQIAHD7T","schema_version":"1.0","canonical_sha256":"1c10038ff329e72949415bba1c96513224f5f9bcde3306d3d741ca157506a7f4","source":{"kind":"arxiv","id":"2204.03031","version":2},"attestation_state":"computed","paper":{"title":"VALUE: Understanding Dialect Disparity in NLU","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Caleb Ziems, Camille Harris, Diyi Yang, Jessica Anderson, Jiaao Chen","submitted_at":"2022-04-06T18:30:56Z","abstract_excerpt":"English Natural Language Understanding (NLU) systems have achieved great performances and even outperformed humans on benchmarks like GLUE and SuperGLUE. However, these benchmarks contain only textbook Standard American English (SAE). Other dialects have been largely overlooked in the NLP community. This leads to biased and inequitable NLU systems that serve only a sub-population of speakers. To understand disparities in current models and to facilitate more dialect-competent NLU systems, we introduce the VernAcular Language Understanding Evaluation (VALUE) benchmark, a challenging variant of "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2204.03031","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-04-06T18:30:56Z","cross_cats_sorted":[],"title_canon_sha256":"43515d1c828a3130bdce44de013590b30a1ba74b03ccbfce316082870ea49b96","abstract_canon_sha256":"75d7faefc765318213962d6c2e7db5f0bf2d03d55c51800fad3f350764c68047"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:56:35.351046Z","signature_b64":"D7lsHwr158dYJEW08AikfoJgRKJjsz6YJzBWLqPx9aplCw17BJiMdk+7o6v3NXHrvPxUTtOLI6T2LVjdyTZ7Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1c10038ff329e72949415bba1c96513224f5f9bcde3306d3d741ca157506a7f4","last_reissued_at":"2026-07-05T04:56:35.350685Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:56:35.350685Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VALUE: Understanding Dialect Disparity in NLU","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Caleb Ziems, Camille Harris, Diyi Yang, Jessica Anderson, Jiaao Chen","submitted_at":"2022-04-06T18:30:56Z","abstract_excerpt":"English Natural Language Understanding (NLU) systems have achieved great performances and even outperformed humans on benchmarks like GLUE and SuperGLUE. However, these benchmarks contain only textbook Standard American English (SAE). Other dialects have been largely overlooked in the NLP community. This leads to biased and inequitable NLU systems that serve only a sub-population of speakers. To understand disparities in current models and to facilitate more dialect-competent NLU systems, we introduce the VernAcular Language Understanding Evaluation (VALUE) benchmark, a challenging variant of "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2204.03031","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.03031/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2204.03031","created_at":"2026-07-05T04:56:35.350749+00:00"},{"alias_kind":"arxiv_version","alias_value":"2204.03031v2","created_at":"2026-07-05T04:56:35.350749+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.03031","created_at":"2026-07-05T04:56:35.350749+00:00"},{"alias_kind":"pith_short_12","alias_value":"DQIAHD7TFHTS","created_at":"2026-07-05T04:56:35.350749+00:00"},{"alias_kind":"pith_short_16","alias_value":"DQIAHD7TFHTSSSKB","created_at":"2026-07-05T04:56:35.350749+00:00"},{"alias_kind":"pith_short_8","alias_value":"DQIAHD7T","created_at":"2026-07-05T04:56:35.350749+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.07058","citing_title":"Using Contextually Aligned Online Reviews to Measure LLMs' Performance Disparities Across Language Varieties","ref_index":32,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DQIAHD7TFHTSSSKBLO5BZFSRGI","json":"https://pith.science/pith/DQIAHD7TFHTSSSKBLO5BZFSRGI.json","graph_json":"https://pith.science/api/pith-number/DQIAHD7TFHTSSSKBLO5BZFSRGI/graph.json","events_json":"https://pith.science/api/pith-number/DQIAHD7TFHTSSSKBLO5BZFSRGI/events.json","paper":"https://pith.science/paper/DQIAHD7T"},"agent_actions":{"view_html":"https://pith.science/pith/DQIAHD7TFHTSSSKBLO5BZFSRGI","download_json":"https://pith.science/pith/DQIAHD7TFHTSSSKBLO5BZFSRGI.json","view_paper":"https://pith.science/paper/DQIAHD7T","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2204.03031&json=true","fetch_graph":"https://pith.science/api/pith-number/DQIAHD7TFHTSSSKBLO5BZFSRGI/graph.json","fetch_events":"https://pith.science/api/pith-number/DQIAHD7TFHTSSSKBLO5BZFSRGI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DQIAHD7TFHTSSSKBLO5BZFSRGI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DQIAHD7TFHTSSSKBLO5BZFSRGI/action/storage_attestation","attest_author":"https://pith.science/pith/DQIAHD7TFHTSSSKBLO5BZFSRGI/action/author_attestation","sign_citation":"https://pith.science/pith/DQIAHD7TFHTSSSKBLO5BZFSRGI/action/citation_signature","submit_replication":"https://pith.science/pith/DQIAHD7TFHTSSSKBLO5BZFSRGI/action/replication_record"}},"created_at":"2026-07-05T04:56:35.350749+00:00","updated_at":"2026-07-05T04:56:35.350749+00:00"}