{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:QEVJQLS4B22JEYNKVMJTX2YIRV","short_pith_number":"pith:QEVJQLS4","schema_version":"1.0","canonical_sha256":"812a982e5c0eb49261aaab133beb088d68e22cd28a0b579af0b979a2e980005a","source":{"kind":"arxiv","id":"2207.10524","version":2},"attestation_state":"computed","paper":{"title":"NusaCrowd: A Call for Open and Reproducible NLP Research in Indonesian Languages","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Ade Romadhony, Alham Fikri Aji, Ayu Purwarianti, Bryan Wilie, David Moeljadi, Fajri Koto, Genta Indra Winata, Holy Lovenia, Karissa Vincentio, Rahmad Mahendra, Samuel Cahyawijaya","submitted_at":"2022-07-21T15:05:42Z","abstract_excerpt":"At the center of the underlying issues that halt Indonesian natural language processing (NLP) research advancement, we find data scarcity. Resources in Indonesian languages, especially the local ones, are extremely scarce and underrepresented. Many Indonesian researchers do not publish their dataset. Furthermore, the few public datasets that we have are scattered across different platforms, thus makes performing reproducible and data-centric research in Indonesian NLP even more arduous. Rising to this challenge, we initiate the first Indonesian NLP crowdsourcing effort, NusaCrowd. NusaCrowd st"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2207.10524","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2022-07-21T15:05:42Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"c14870686027312cfc8b550c5317544fc58306e5133c222029b3070b2564cdf7","abstract_canon_sha256":"8a20160e4660b605d642ed43a7a969fc26f1799e53d9e1172840539df629dad8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:45:01.897718Z","signature_b64":"EGre90YYrIYjp6Oabos7xrzXbAzEVe8//XxKPQG2JGDaMD6N4uZH33MTcaIH4OtG3/al8GvjfaUpW1DdKLh1Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"812a982e5c0eb49261aaab133beb088d68e22cd28a0b579af0b979a2e980005a","last_reissued_at":"2026-07-05T04:45:01.897264Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:45:01.897264Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"NusaCrowd: A Call for Open and Reproducible NLP Research in Indonesian Languages","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Ade Romadhony, Alham Fikri Aji, Ayu Purwarianti, Bryan Wilie, David Moeljadi, Fajri Koto, Genta Indra Winata, Holy Lovenia, Karissa Vincentio, Rahmad Mahendra, Samuel Cahyawijaya","submitted_at":"2022-07-21T15:05:42Z","abstract_excerpt":"At the center of the underlying issues that halt Indonesian natural language processing (NLP) research advancement, we find data scarcity. Resources in Indonesian languages, especially the local ones, are extremely scarce and underrepresented. Many Indonesian researchers do not publish their dataset. Furthermore, the few public datasets that we have are scattered across different platforms, thus makes performing reproducible and data-centric research in Indonesian NLP even more arduous. Rising to this challenge, we initiate the first Indonesian NLP crowdsourcing effort, NusaCrowd. NusaCrowd st"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2207.10524","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2207.10524/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2207.10524","created_at":"2026-07-05T04:45:01.897323+00:00"},{"alias_kind":"arxiv_version","alias_value":"2207.10524v2","created_at":"2026-07-05T04:45:01.897323+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2207.10524","created_at":"2026-07-05T04:45:01.897323+00:00"},{"alias_kind":"pith_short_12","alias_value":"QEVJQLS4B22J","created_at":"2026-07-05T04:45:01.897323+00:00"},{"alias_kind":"pith_short_16","alias_value":"QEVJQLS4B22JEYNK","created_at":"2026-07-05T04:45:01.897323+00:00"},{"alias_kind":"pith_short_8","alias_value":"QEVJQLS4","created_at":"2026-07-05T04:45:01.897323+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QEVJQLS4B22JEYNKVMJTX2YIRV","json":"https://pith.science/pith/QEVJQLS4B22JEYNKVMJTX2YIRV.json","graph_json":"https://pith.science/api/pith-number/QEVJQLS4B22JEYNKVMJTX2YIRV/graph.json","events_json":"https://pith.science/api/pith-number/QEVJQLS4B22JEYNKVMJTX2YIRV/events.json","paper":"https://pith.science/paper/QEVJQLS4"},"agent_actions":{"view_html":"https://pith.science/pith/QEVJQLS4B22JEYNKVMJTX2YIRV","download_json":"https://pith.science/pith/QEVJQLS4B22JEYNKVMJTX2YIRV.json","view_paper":"https://pith.science/paper/QEVJQLS4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2207.10524&json=true","fetch_graph":"https://pith.science/api/pith-number/QEVJQLS4B22JEYNKVMJTX2YIRV/graph.json","fetch_events":"https://pith.science/api/pith-number/QEVJQLS4B22JEYNKVMJTX2YIRV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QEVJQLS4B22JEYNKVMJTX2YIRV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QEVJQLS4B22JEYNKVMJTX2YIRV/action/storage_attestation","attest_author":"https://pith.science/pith/QEVJQLS4B22JEYNKVMJTX2YIRV/action/author_attestation","sign_citation":"https://pith.science/pith/QEVJQLS4B22JEYNKVMJTX2YIRV/action/citation_signature","submit_replication":"https://pith.science/pith/QEVJQLS4B22JEYNKVMJTX2YIRV/action/replication_record"}},"created_at":"2026-07-05T04:45:01.897323+00:00","updated_at":"2026-07-05T04:45:01.897323+00:00"}