{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:C33J3F5HBF7H245BXBEKGCZT6F","short_pith_number":"pith:C33J3F5H","schema_version":"1.0","canonical_sha256":"16f69d97a7097e7d73a1b848a30b33f15a71b8c36073247a7bd9df306b4828d5","source":{"kind":"arxiv","id":"2208.01448","version":2},"attestation_state":"computed","paper":{"title":"AlexaTM 20B: Few-Shot Learning Using a Large-Scale Multilingual Seq2Seq Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Andy Rosenbaum, Anna Rumshisky, Apurv Verma, Chandana Satya Prakash, Charith Peris, Fabian Triefenbach, Gokhan Tur, Haidar Khan, Jack FitzGerald, Mukund Sridhar, Prem Natarajan, Rahul Gupta, Saleh Soltan, Shankar Ananthakrishnan, Stephen Rawls, Wael Hamza","submitted_at":"2022-08-02T13:30:07Z","abstract_excerpt":"In this work, we demonstrate that multilingual large-scale sequence-to-sequence (seq2seq) models, pre-trained on a mixture of denoising and Causal Language Modeling (CLM) tasks, are more efficient few-shot learners than decoder-only models on various tasks. In particular, we train a 20 billion parameter multilingual seq2seq model called Alexa Teacher Model (AlexaTM 20B) and show that it achieves state-of-the-art (SOTA) performance on 1-shot summarization tasks, outperforming a much larger 540B PaLM decoder model. AlexaTM 20B also achieves SOTA in 1-shot machine translation, especially for low-"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2208.01448","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2022-08-02T13:30:07Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"cee22d25dc272bafa84952745bf7b8cc6e93d1f305a6e364291d302ce5d3e6b1","abstract_canon_sha256":"fdbebea2e8a9da3694361e9700d167b6f9990ef055957ed0eb1188b11b7b7837"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:45:50.155785Z","signature_b64":"ZycbGbexUrLQgWrzukB4emlX87Oye+p14BzxS5CrnH7WR7kjTeMjucm4BTpWADy1Me2JBZN4Jy+bf5/LuO8XDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"16f69d97a7097e7d73a1b848a30b33f15a71b8c36073247a7bd9df306b4828d5","last_reissued_at":"2026-07-05T04:45:50.155345Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:45:50.155345Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AlexaTM 20B: Few-Shot Learning Using a Large-Scale Multilingual Seq2Seq Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Andy Rosenbaum, Anna Rumshisky, Apurv Verma, Chandana Satya Prakash, Charith Peris, Fabian Triefenbach, Gokhan Tur, Haidar Khan, Jack FitzGerald, Mukund Sridhar, Prem Natarajan, Rahul Gupta, Saleh Soltan, Shankar Ananthakrishnan, Stephen Rawls, Wael Hamza","submitted_at":"2022-08-02T13:30:07Z","abstract_excerpt":"In this work, we demonstrate that multilingual large-scale sequence-to-sequence (seq2seq) models, pre-trained on a mixture of denoising and Causal Language Modeling (CLM) tasks, are more efficient few-shot learners than decoder-only models on various tasks. In particular, we train a 20 billion parameter multilingual seq2seq model called Alexa Teacher Model (AlexaTM 20B) and show that it achieves state-of-the-art (SOTA) performance on 1-shot summarization tasks, outperforming a much larger 540B PaLM decoder model. AlexaTM 20B also achieves SOTA in 1-shot machine translation, especially for low-"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2208.01448","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2208.01448/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2208.01448","created_at":"2026-07-05T04:45:50.155399+00:00"},{"alias_kind":"arxiv_version","alias_value":"2208.01448v2","created_at":"2026-07-05T04:45:50.155399+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2208.01448","created_at":"2026-07-05T04:45:50.155399+00:00"},{"alias_kind":"pith_short_12","alias_value":"C33J3F5HBF7H","created_at":"2026-07-05T04:45:50.155399+00:00"},{"alias_kind":"pith_short_16","alias_value":"C33J3F5HBF7H245B","created_at":"2026-07-05T04:45:50.155399+00:00"},{"alias_kind":"pith_short_8","alias_value":"C33J3F5H","created_at":"2026-07-05T04:45:50.155399+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2307.06435","citing_title":"A Comprehensive Overview of Large Language Models","ref_index":122,"is_internal_anchor":false},{"citing_arxiv_id":"2305.07922","citing_title":"CodeT5+: Open Code Large Language Models for Code Understanding and Generation","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2305.16264","citing_title":"Scaling Data-Constrained Language Models","ref_index":108,"is_internal_anchor":false},{"citing_arxiv_id":"2303.17564","citing_title":"BloombergGPT: A Large Language Model for Finance","ref_index":106,"is_internal_anchor":false},{"citing_arxiv_id":"2211.05100","citing_title":"BLOOM: A 176B-Parameter Open-Access Multilingual Language Model","ref_index":140,"is_internal_anchor":false},{"citing_arxiv_id":"2402.06196","citing_title":"Large Language Models: A Survey","ref_index":89,"is_internal_anchor":false},{"citing_arxiv_id":"2303.18223","citing_title":"A Survey of Large Language Models","ref_index":117,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02300","citing_title":"A Meta Reinforcement Learning Approach to Goals-Based Wealth Management","ref_index":244,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/C33J3F5HBF7H245BXBEKGCZT6F","json":"https://pith.science/pith/C33J3F5HBF7H245BXBEKGCZT6F.json","graph_json":"https://pith.science/api/pith-number/C33J3F5HBF7H245BXBEKGCZT6F/graph.json","events_json":"https://pith.science/api/pith-number/C33J3F5HBF7H245BXBEKGCZT6F/events.json","paper":"https://pith.science/paper/C33J3F5H"},"agent_actions":{"view_html":"https://pith.science/pith/C33J3F5HBF7H245BXBEKGCZT6F","download_json":"https://pith.science/pith/C33J3F5HBF7H245BXBEKGCZT6F.json","view_paper":"https://pith.science/paper/C33J3F5H","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2208.01448&json=true","fetch_graph":"https://pith.science/api/pith-number/C33J3F5HBF7H245BXBEKGCZT6F/graph.json","fetch_events":"https://pith.science/api/pith-number/C33J3F5HBF7H245BXBEKGCZT6F/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/C33J3F5HBF7H245BXBEKGCZT6F/action/timestamp_anchor","attest_storage":"https://pith.science/pith/C33J3F5HBF7H245BXBEKGCZT6F/action/storage_attestation","attest_author":"https://pith.science/pith/C33J3F5HBF7H245BXBEKGCZT6F/action/author_attestation","sign_citation":"https://pith.science/pith/C33J3F5HBF7H245BXBEKGCZT6F/action/citation_signature","submit_replication":"https://pith.science/pith/C33J3F5HBF7H245BXBEKGCZT6F/action/replication_record"}},"created_at":"2026-07-05T04:45:50.155399+00:00","updated_at":"2026-07-05T04:45:50.155399+00:00"}