{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:QQ3PWKE7WID3WWFPWLBXS3TBS2","short_pith_number":"pith:QQ3PWKE7","schema_version":"1.0","canonical_sha256":"8436fb289fb207bb58afb2c3796e6196b50b9c3180d211e421970fa33cbab42e","source":{"kind":"arxiv","id":"2502.07058","version":3},"attestation_state":"computed","paper":{"title":"Using Contextually Aligned Online Reviews to Measure LLMs' Performance Disparities Across Language Varieties","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.HC"],"primary_cat":"cs.CL","authors_text":"Chieh-Yang Huang, Hen-Hsen Huang, Ho Yin Sam Ng, Ting-Hao 'Kenneth' Huang, Tsung-Che Li, Zixin Tang","submitted_at":"2025-02-10T21:49:35Z","abstract_excerpt":"A language can have different varieties. These varieties can affect the performance of natural language processing (NLP) models, including large language models (LLMs), which are often trained on data from widely spoken varieties. This paper introduces a novel and cost-effective approach to benchmark model performance across language varieties. We argue that international online review platforms, such as Booking.com, can serve as effective data sources for constructing datasets that capture comments in different language varieties from similar real-world scenarios, like reviews for the same ho"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.07058","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-10T21:49:35Z","cross_cats_sorted":["cs.HC"],"title_canon_sha256":"7904dc13fe89e7bd343669caa9581cd3099b6defdc1eba2fb4030668e3e3c14c","abstract_canon_sha256":"d266ed6772d5176594e5dacf74a6721e1169b6428a7efd139533c213ff36a18d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:35:46.218661Z","signature_b64":"1rd9e/9pvAIqVP4063yMvo1QLSgQOkJnyDXeCNyh4Gv5cKPQsiz/xT9fUbyD8+zaPsNddvu0jW8bL1PAtEKnAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8436fb289fb207bb58afb2c3796e6196b50b9c3180d211e421970fa33cbab42e","last_reissued_at":"2026-07-05T10:35:46.217852Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:35:46.217852Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Using Contextually Aligned Online Reviews to Measure LLMs' Performance Disparities Across Language Varieties","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.HC"],"primary_cat":"cs.CL","authors_text":"Chieh-Yang Huang, Hen-Hsen Huang, Ho Yin Sam Ng, Ting-Hao 'Kenneth' Huang, Tsung-Che Li, Zixin Tang","submitted_at":"2025-02-10T21:49:35Z","abstract_excerpt":"A language can have different varieties. These varieties can affect the performance of natural language processing (NLP) models, including large language models (LLMs), which are often trained on data from widely spoken varieties. This paper introduces a novel and cost-effective approach to benchmark model performance across language varieties. We argue that international online review platforms, such as Booking.com, can serve as effective data sources for constructing datasets that capture comments in different language varieties from similar real-world scenarios, like reviews for the same ho"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.07058","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.07058/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.07058","created_at":"2026-07-05T10:35:46.217953+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.07058v3","created_at":"2026-07-05T10:35:46.217953+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.07058","created_at":"2026-07-05T10:35:46.217953+00:00"},{"alias_kind":"pith_short_12","alias_value":"QQ3PWKE7WID3","created_at":"2026-07-05T10:35:46.217953+00:00"},{"alias_kind":"pith_short_16","alias_value":"QQ3PWKE7WID3WWFP","created_at":"2026-07-05T10:35:46.217953+00:00"},{"alias_kind":"pith_short_8","alias_value":"QQ3PWKE7","created_at":"2026-07-05T10:35:46.217953+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QQ3PWKE7WID3WWFPWLBXS3TBS2","json":"https://pith.science/pith/QQ3PWKE7WID3WWFPWLBXS3TBS2.json","graph_json":"https://pith.science/api/pith-number/QQ3PWKE7WID3WWFPWLBXS3TBS2/graph.json","events_json":"https://pith.science/api/pith-number/QQ3PWKE7WID3WWFPWLBXS3TBS2/events.json","paper":"https://pith.science/paper/QQ3PWKE7"},"agent_actions":{"view_html":"https://pith.science/pith/QQ3PWKE7WID3WWFPWLBXS3TBS2","download_json":"https://pith.science/pith/QQ3PWKE7WID3WWFPWLBXS3TBS2.json","view_paper":"https://pith.science/paper/QQ3PWKE7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.07058&json=true","fetch_graph":"https://pith.science/api/pith-number/QQ3PWKE7WID3WWFPWLBXS3TBS2/graph.json","fetch_events":"https://pith.science/api/pith-number/QQ3PWKE7WID3WWFPWLBXS3TBS2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QQ3PWKE7WID3WWFPWLBXS3TBS2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QQ3PWKE7WID3WWFPWLBXS3TBS2/action/storage_attestation","attest_author":"https://pith.science/pith/QQ3PWKE7WID3WWFPWLBXS3TBS2/action/author_attestation","sign_citation":"https://pith.science/pith/QQ3PWKE7WID3WWFPWLBXS3TBS2/action/citation_signature","submit_replication":"https://pith.science/pith/QQ3PWKE7WID3WWFPWLBXS3TBS2/action/replication_record"}},"created_at":"2026-07-05T10:35:46.217953+00:00","updated_at":"2026-07-05T10:35:46.217953+00:00"}