{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:X2ZNCUPEXO4VPOAA5M7M2S2VPL","short_pith_number":"pith:X2ZNCUPE","schema_version":"1.0","canonical_sha256":"beb2d151e4bbb957b800eb3ecd4b557ae2ca16bd12441bcb024515cc6602fcdb","source":{"kind":"arxiv","id":"2203.13357","version":1},"attestation_state":"computed","paper":{"title":"One Country, 700+ Languages: NLP Challenges for Underrepresented Languages and Dialects in Indonesia","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ade Romadhony, Alham Fikri Aji, David Moeljadi, Fajri Koto, Genta Indra Winata, Jey Han Lau, Kemal Kurniawan, Radityo Eko Prasojo, Rahmad Mahendra, Samuel Cahyawijaya, Sebastian Ruder, Timothy Baldwin","submitted_at":"2022-03-24T22:07:22Z","abstract_excerpt":"NLP research is impeded by a lack of resources and awareness of the challenges presented by underrepresented languages and dialects. Focusing on the languages spoken in Indonesia, the second most linguistically diverse and the fourth most populous nation of the world, we provide an overview of the current state of NLP research for Indonesia's 700+ languages. We highlight challenges in Indonesian NLP and how these affect the performance of current NLP systems. Finally, we provide general recommendations to help develop NLP technology not only for languages of Indonesia but also other underrepre"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2203.13357","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2022-03-24T22:07:22Z","cross_cats_sorted":[],"title_canon_sha256":"880f1822985620438b3f25d9b36d169b7451e8f5a89c97ca7cc5c7dce334432a","abstract_canon_sha256":"83a1eded9223bcfe0ac1118812fdb9f20d5d177f07b33305d6fd5d3d1452a4b4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:08:31.054300Z","signature_b64":"PbLbWrM/4NW6QWzMrffEt81RN9T/U5nicHeErNTRXs9XTtiNm2ugsee7dgUt2Ff+lSTPLMPEAa2sc29y9aZMAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"beb2d151e4bbb957b800eb3ecd4b557ae2ca16bd12441bcb024515cc6602fcdb","last_reissued_at":"2026-07-05T04:08:31.053838Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:08:31.053838Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"One Country, 700+ Languages: NLP Challenges for Underrepresented Languages and Dialects in Indonesia","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ade Romadhony, Alham Fikri Aji, David Moeljadi, Fajri Koto, Genta Indra Winata, Jey Han Lau, Kemal Kurniawan, Radityo Eko Prasojo, Rahmad Mahendra, Samuel Cahyawijaya, Sebastian Ruder, Timothy Baldwin","submitted_at":"2022-03-24T22:07:22Z","abstract_excerpt":"NLP research is impeded by a lack of resources and awareness of the challenges presented by underrepresented languages and dialects. Focusing on the languages spoken in Indonesia, the second most linguistically diverse and the fourth most populous nation of the world, we provide an overview of the current state of NLP research for Indonesia's 700+ languages. We highlight challenges in Indonesian NLP and how these affect the performance of current NLP systems. Finally, we provide general recommendations to help develop NLP technology not only for languages of Indonesia but also other underrepre"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.13357","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.13357/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2203.13357","created_at":"2026-07-05T04:08:31.053905+00:00"},{"alias_kind":"arxiv_version","alias_value":"2203.13357v1","created_at":"2026-07-05T04:08:31.053905+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.13357","created_at":"2026-07-05T04:08:31.053905+00:00"},{"alias_kind":"pith_short_12","alias_value":"X2ZNCUPEXO4V","created_at":"2026-07-05T04:08:31.053905+00:00"},{"alias_kind":"pith_short_16","alias_value":"X2ZNCUPEXO4VPOAA","created_at":"2026-07-05T04:08:31.053905+00:00"},{"alias_kind":"pith_short_8","alias_value":"X2ZNCUPE","created_at":"2026-07-05T04:08:31.053905+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.09943","citing_title":"Indigenous Languages Spoken in Argentina: A Survey of NLP and Speech Resources","ref_index":12,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/X2ZNCUPEXO4VPOAA5M7M2S2VPL","json":"https://pith.science/pith/X2ZNCUPEXO4VPOAA5M7M2S2VPL.json","graph_json":"https://pith.science/api/pith-number/X2ZNCUPEXO4VPOAA5M7M2S2VPL/graph.json","events_json":"https://pith.science/api/pith-number/X2ZNCUPEXO4VPOAA5M7M2S2VPL/events.json","paper":"https://pith.science/paper/X2ZNCUPE"},"agent_actions":{"view_html":"https://pith.science/pith/X2ZNCUPEXO4VPOAA5M7M2S2VPL","download_json":"https://pith.science/pith/X2ZNCUPEXO4VPOAA5M7M2S2VPL.json","view_paper":"https://pith.science/paper/X2ZNCUPE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2203.13357&json=true","fetch_graph":"https://pith.science/api/pith-number/X2ZNCUPEXO4VPOAA5M7M2S2VPL/graph.json","fetch_events":"https://pith.science/api/pith-number/X2ZNCUPEXO4VPOAA5M7M2S2VPL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/X2ZNCUPEXO4VPOAA5M7M2S2VPL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/X2ZNCUPEXO4VPOAA5M7M2S2VPL/action/storage_attestation","attest_author":"https://pith.science/pith/X2ZNCUPEXO4VPOAA5M7M2S2VPL/action/author_attestation","sign_citation":"https://pith.science/pith/X2ZNCUPEXO4VPOAA5M7M2S2VPL/action/citation_signature","submit_replication":"https://pith.science/pith/X2ZNCUPEXO4VPOAA5M7M2S2VPL/action/replication_record"}},"created_at":"2026-07-05T04:08:31.053905+00:00","updated_at":"2026-07-05T04:08:31.053905+00:00"}