{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:JEEE7YFM6VT2MR7LIMT6RHRU46","short_pith_number":"pith:JEEE7YFM","schema_version":"1.0","canonical_sha256":"49084fe0acf567a647eb4327e89e34e78cac225fb1fbf41b0d7b08f36b58e2ad","source":{"kind":"arxiv","id":"2305.13641","version":1},"attestation_state":"computed","paper":{"title":"AxomiyaBERTa: A Phonologically-aware Transformer Model for Assamese","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Abhijnan Nath, Nikhil Krishnaswamy, Sheikh Mannan","submitted_at":"2023-05-23T03:19:21Z","abstract_excerpt":"Despite their successes in NLP, Transformer-based language models still require extensive computing resources and suffer in low-resource or low-compute settings. In this paper, we present AxomiyaBERTa, a novel BERT model for Assamese, a morphologically-rich low-resource language (LRL) of Eastern India. AxomiyaBERTa is trained only on the masked language modeling (MLM) task, without the typical additional next sentence prediction (NSP) objective, and our results show that in resource-scarce settings for very low-resource languages like Assamese, MLM alone can be successfully leveraged for a ran"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.13641","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-05-23T03:19:21Z","cross_cats_sorted":[],"title_canon_sha256":"196d46bd767fdeb216e61616f93cc11d424e4dee846b70710072ff008c46892d","abstract_canon_sha256":"41f33b441081c2633126b888034bf53efa524d6b661e73170c848324212fe549"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:12:55.441490Z","signature_b64":"SDPCPqqunR6KbDJq81PE4aVi/OQ8XMvmiIjqvtvEyblIGSjLTj5z+USZbGm4kcJT+savEcSck43BD/c6VRP6Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"49084fe0acf567a647eb4327e89e34e78cac225fb1fbf41b0d7b08f36b58e2ad","last_reissued_at":"2026-07-05T06:12:55.440969Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:12:55.440969Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AxomiyaBERTa: A Phonologically-aware Transformer Model for Assamese","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Abhijnan Nath, Nikhil Krishnaswamy, Sheikh Mannan","submitted_at":"2023-05-23T03:19:21Z","abstract_excerpt":"Despite their successes in NLP, Transformer-based language models still require extensive computing resources and suffer in low-resource or low-compute settings. In this paper, we present AxomiyaBERTa, a novel BERT model for Assamese, a morphologically-rich low-resource language (LRL) of Eastern India. AxomiyaBERTa is trained only on the masked language modeling (MLM) task, without the typical additional next sentence prediction (NSP) objective, and our results show that in resource-scarce settings for very low-resource languages like Assamese, MLM alone can be successfully leveraged for a ran"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.13641","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.13641/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.13641","created_at":"2026-07-05T06:12:55.441032+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.13641v1","created_at":"2026-07-05T06:12:55.441032+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.13641","created_at":"2026-07-05T06:12:55.441032+00:00"},{"alias_kind":"pith_short_12","alias_value":"JEEE7YFM6VT2","created_at":"2026-07-05T06:12:55.441032+00:00"},{"alias_kind":"pith_short_16","alias_value":"JEEE7YFM6VT2MR7L","created_at":"2026-07-05T06:12:55.441032+00:00"},{"alias_kind":"pith_short_8","alias_value":"JEEE7YFM","created_at":"2026-07-05T06:12:55.441032+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.00029","citing_title":"A Breadth-First Catalog of Text Processing, Speech Processing and Multimodal Research in South Asian Languages","ref_index":12,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JEEE7YFM6VT2MR7LIMT6RHRU46","json":"https://pith.science/pith/JEEE7YFM6VT2MR7LIMT6RHRU46.json","graph_json":"https://pith.science/api/pith-number/JEEE7YFM6VT2MR7LIMT6RHRU46/graph.json","events_json":"https://pith.science/api/pith-number/JEEE7YFM6VT2MR7LIMT6RHRU46/events.json","paper":"https://pith.science/paper/JEEE7YFM"},"agent_actions":{"view_html":"https://pith.science/pith/JEEE7YFM6VT2MR7LIMT6RHRU46","download_json":"https://pith.science/pith/JEEE7YFM6VT2MR7LIMT6RHRU46.json","view_paper":"https://pith.science/paper/JEEE7YFM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.13641&json=true","fetch_graph":"https://pith.science/api/pith-number/JEEE7YFM6VT2MR7LIMT6RHRU46/graph.json","fetch_events":"https://pith.science/api/pith-number/JEEE7YFM6VT2MR7LIMT6RHRU46/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JEEE7YFM6VT2MR7LIMT6RHRU46/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JEEE7YFM6VT2MR7LIMT6RHRU46/action/storage_attestation","attest_author":"https://pith.science/pith/JEEE7YFM6VT2MR7LIMT6RHRU46/action/author_attestation","sign_citation":"https://pith.science/pith/JEEE7YFM6VT2MR7LIMT6RHRU46/action/citation_signature","submit_replication":"https://pith.science/pith/JEEE7YFM6VT2MR7LIMT6RHRU46/action/replication_record"}},"created_at":"2026-07-05T06:12:55.441032+00:00","updated_at":"2026-07-05T06:12:55.441032+00:00"}