{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:IL64DHSMXWGTG6C4UBCJAI6JN7","short_pith_number":"pith:IL64DHSM","schema_version":"1.0","canonical_sha256":"42fdc19e4cbd8d33785ca0449023c96fe7d7e73a4f3643aee4b4779fc9a2ea76","source":{"kind":"arxiv","id":"2308.16149","version":2},"attestation_state":"computed","paper":{"title":"Jais and Jais-chat: Arabic-Centric Foundation and Instruction-Tuned Open Generative Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Alham Fikri Aji, Andrew Feldman, Andrew Jackson, Andy Hock, Bokang Jia, Cynthia Liu, Eric Xing, Fajri Koto, Gurpreet Gosal, Haonan Li, Hector Xuguang Ren, Joel Hestness, Jonathan Lee, Lalit Pradhan, Massa Baali, Natalia Vassilieva, Neha Sengupta, Onkar Pandit, Osama Mohammed Afzal, Preslav Nakov, Rahul Pal, Samta Kamboj, Satheesh Katipomu, Sondos Mahmoud Bsharat, Sunil Kumar Sahu, Timothy Baldwin, William Marshall, Xudong Han, Zain Muhammad Mujahid, Zhengzhong Liu, Zhiming Chen, Zhiqiang Shen","submitted_at":"2023-08-30T17:07:17Z","abstract_excerpt":"We introduce Jais and Jais-chat, new state-of-the-art Arabic-centric foundation and instruction-tuned open generative large language models (LLMs). The models are based on the GPT-3 decoder-only architecture and are pretrained on a mixture of Arabic and English texts, including source code in various programming languages. With 13 billion parameters, they demonstrate better knowledge and reasoning capabilities in Arabic than any existing open Arabic and multilingual models by a sizable margin, based on extensive evaluation. Moreover, the models are competitive in English compared to English-ce"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.16149","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-08-30T17:07:17Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"b8455a1bd3fffb6cd491a29df99c4ac4435ac98765e52d45b99eecfeccffd6ed","abstract_canon_sha256":"c2e64f5010b4f02f9d03ad56b5c67c998afca3bb449161dd8bed8d4cf31faf61"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:55:40.386026Z","signature_b64":"6mhqjw96ylJCTR8t9KTHp8FLN+2ZRygAKXoB4fQs3WopOU0k1KzdEiuc5UR6iYCLWMnSvmqkqJshYWg9nL7xBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"42fdc19e4cbd8d33785ca0449023c96fe7d7e73a4f3643aee4b4779fc9a2ea76","last_reissued_at":"2026-07-05T06:55:40.385550Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:55:40.385550Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Jais and Jais-chat: Arabic-Centric Foundation and Instruction-Tuned Open Generative Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Alham Fikri Aji, Andrew Feldman, Andrew Jackson, Andy Hock, Bokang Jia, Cynthia Liu, Eric Xing, Fajri Koto, Gurpreet Gosal, Haonan Li, Hector Xuguang Ren, Joel Hestness, Jonathan Lee, Lalit Pradhan, Massa Baali, Natalia Vassilieva, Neha Sengupta, Onkar Pandit, Osama Mohammed Afzal, Preslav Nakov, Rahul Pal, Samta Kamboj, Satheesh Katipomu, Sondos Mahmoud Bsharat, Sunil Kumar Sahu, Timothy Baldwin, William Marshall, Xudong Han, Zain Muhammad Mujahid, Zhengzhong Liu, Zhiming Chen, Zhiqiang Shen","submitted_at":"2023-08-30T17:07:17Z","abstract_excerpt":"We introduce Jais and Jais-chat, new state-of-the-art Arabic-centric foundation and instruction-tuned open generative large language models (LLMs). The models are based on the GPT-3 decoder-only architecture and are pretrained on a mixture of Arabic and English texts, including source code in various programming languages. With 13 billion parameters, they demonstrate better knowledge and reasoning capabilities in Arabic than any existing open Arabic and multilingual models by a sizable margin, based on extensive evaluation. Moreover, the models are competitive in English compared to English-ce"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.16149","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.16149/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.16149","created_at":"2026-07-05T06:55:40.385607+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.16149v2","created_at":"2026-07-05T06:55:40.385607+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.16149","created_at":"2026-07-05T06:55:40.385607+00:00"},{"alias_kind":"pith_short_12","alias_value":"IL64DHSMXWGT","created_at":"2026-07-05T06:55:40.385607+00:00"},{"alias_kind":"pith_short_16","alias_value":"IL64DHSMXWGTG6C4","created_at":"2026-07-05T06:55:40.385607+00:00"},{"alias_kind":"pith_short_8","alias_value":"IL64DHSM","created_at":"2026-07-05T06:55:40.385607+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22841","citing_title":"IndicGuard: A Multilingual Safety Guard Model and Dataset for Indic Languages","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10765","citing_title":"ArabiGEE: A Hierarchical Taxonomy for Arabic Grammatical Error Explanation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07422","citing_title":"The Masked Advantage: Uncovering Local-Language Access to Cultural Knowledge in LLMs","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19714","citing_title":"LLM-Based Financial Sentiment Analysis in Arabic: Evidence from Saudi Markets","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17007","citing_title":"HalluScore: Large Language Model Hallucination Question Answering Benchmark","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2310.02446","citing_title":"Low-Resource Languages Jailbreak GPT-4","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03380","citing_title":"Noise Steering for Controlled Text Generation: Improving Diversity and Reading-Level Fidelity in Arabic Educational Story Generation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05662","citing_title":"XL-SafetyBench: A Country-Grounded Cross-Cultural Benchmark for LLM Safety and Cultural Sensitivity","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00119","citing_title":"Cultural Benchmarking of LLMs in Standard and Dialectal Arabic Dialogues","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18490","citing_title":"LQM: Linguistically Motivated Multidimensional Quality Metrics for Machine Translation","ref_index":91,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IL64DHSMXWGTG6C4UBCJAI6JN7","json":"https://pith.science/pith/IL64DHSMXWGTG6C4UBCJAI6JN7.json","graph_json":"https://pith.science/api/pith-number/IL64DHSMXWGTG6C4UBCJAI6JN7/graph.json","events_json":"https://pith.science/api/pith-number/IL64DHSMXWGTG6C4UBCJAI6JN7/events.json","paper":"https://pith.science/paper/IL64DHSM"},"agent_actions":{"view_html":"https://pith.science/pith/IL64DHSMXWGTG6C4UBCJAI6JN7","download_json":"https://pith.science/pith/IL64DHSMXWGTG6C4UBCJAI6JN7.json","view_paper":"https://pith.science/paper/IL64DHSM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.16149&json=true","fetch_graph":"https://pith.science/api/pith-number/IL64DHSMXWGTG6C4UBCJAI6JN7/graph.json","fetch_events":"https://pith.science/api/pith-number/IL64DHSMXWGTG6C4UBCJAI6JN7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IL64DHSMXWGTG6C4UBCJAI6JN7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IL64DHSMXWGTG6C4UBCJAI6JN7/action/storage_attestation","attest_author":"https://pith.science/pith/IL64DHSMXWGTG6C4UBCJAI6JN7/action/author_attestation","sign_citation":"https://pith.science/pith/IL64DHSMXWGTG6C4UBCJAI6JN7/action/citation_signature","submit_replication":"https://pith.science/pith/IL64DHSMXWGTG6C4UBCJAI6JN7/action/replication_record"}},"created_at":"2026-07-05T06:55:40.385607+00:00","updated_at":"2026-07-05T06:55:40.385607+00:00"}