{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:Q2SVGR4YB7OY7KVP3NJ77FHDI2","short_pith_number":"pith:Q2SVGR4Y","schema_version":"1.0","canonical_sha256":"86a55347980fdd8faaafdb53ff94e346a3bc0d001350b9c29267a51767722a22","source":{"kind":"arxiv","id":"2410.03001","version":1},"attestation_state":"computed","paper":{"title":"Can Transformers Learn $n$-gram Language Models?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Anej Svete, Isabelle Augenstein, Mike Zhou, Nadav Borenstein, Ryan Cotterell","submitted_at":"2024-10-03T21:21:02Z","abstract_excerpt":"Much theoretical work has described the ability of transformers to represent formal languages. However, linking theoretical results to empirical performance is not straightforward due to the complex interplay between the architecture, the learning algorithm, and training data. To test whether theoretical lower bounds imply \\emph{learnability} of formal languages, we turn to recent work relating transformers to $n$-gram language models (LMs). We study transformers' ability to learn random $n$-gram LMs of two kinds: ones with arbitrary next-symbol probabilities and ones where those are defined w"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.03001","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-03T21:21:02Z","cross_cats_sorted":[],"title_canon_sha256":"072722f75662be520084ee5348a25ca02d645b5036419211e5c3f9a7bfd63a3f","abstract_canon_sha256":"ba0273e15a2880cda4d2c73f937b12c599f6c969a192a862c12ac07d9181f4f6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:15:42.929995Z","signature_b64":"6H8HTlt9cXr/WMvOFcHD8n3hsav41LT8Y2E7+JKsi3ntT5iTbuYOGedAeJpbxnB39i10USIK/hQ9BaP/NiyjCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"86a55347980fdd8faaafdb53ff94e346a3bc0d001350b9c29267a51767722a22","last_reissued_at":"2026-07-05T09:15:42.929532Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:15:42.929532Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can Transformers Learn $n$-gram Language Models?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Anej Svete, Isabelle Augenstein, Mike Zhou, Nadav Borenstein, Ryan Cotterell","submitted_at":"2024-10-03T21:21:02Z","abstract_excerpt":"Much theoretical work has described the ability of transformers to represent formal languages. However, linking theoretical results to empirical performance is not straightforward due to the complex interplay between the architecture, the learning algorithm, and training data. To test whether theoretical lower bounds imply \\emph{learnability} of formal languages, we turn to recent work relating transformers to $n$-gram language models (LMs). We study transformers' ability to learn random $n$-gram LMs of two kinds: ones with arbitrary next-symbol probabilities and ones where those are defined w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.03001","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.03001/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.03001","created_at":"2026-07-05T09:15:42.929587+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.03001v1","created_at":"2026-07-05T09:15:42.929587+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.03001","created_at":"2026-07-05T09:15:42.929587+00:00"},{"alias_kind":"pith_short_12","alias_value":"Q2SVGR4YB7OY","created_at":"2026-07-05T09:15:42.929587+00:00"},{"alias_kind":"pith_short_16","alias_value":"Q2SVGR4YB7OY7KVP","created_at":"2026-07-05T09:15:42.929587+00:00"},{"alias_kind":"pith_short_8","alias_value":"Q2SVGR4Y","created_at":"2026-07-05T09:15:42.929587+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.07070","citing_title":"Scaling Laws and Representation Learning in Simple Hierarchical Languages: Transformers vs. Convolutional Architectures","ref_index":45,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Q2SVGR4YB7OY7KVP3NJ77FHDI2","json":"https://pith.science/pith/Q2SVGR4YB7OY7KVP3NJ77FHDI2.json","graph_json":"https://pith.science/api/pith-number/Q2SVGR4YB7OY7KVP3NJ77FHDI2/graph.json","events_json":"https://pith.science/api/pith-number/Q2SVGR4YB7OY7KVP3NJ77FHDI2/events.json","paper":"https://pith.science/paper/Q2SVGR4Y"},"agent_actions":{"view_html":"https://pith.science/pith/Q2SVGR4YB7OY7KVP3NJ77FHDI2","download_json":"https://pith.science/pith/Q2SVGR4YB7OY7KVP3NJ77FHDI2.json","view_paper":"https://pith.science/paper/Q2SVGR4Y","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.03001&json=true","fetch_graph":"https://pith.science/api/pith-number/Q2SVGR4YB7OY7KVP3NJ77FHDI2/graph.json","fetch_events":"https://pith.science/api/pith-number/Q2SVGR4YB7OY7KVP3NJ77FHDI2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Q2SVGR4YB7OY7KVP3NJ77FHDI2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Q2SVGR4YB7OY7KVP3NJ77FHDI2/action/storage_attestation","attest_author":"https://pith.science/pith/Q2SVGR4YB7OY7KVP3NJ77FHDI2/action/author_attestation","sign_citation":"https://pith.science/pith/Q2SVGR4YB7OY7KVP3NJ77FHDI2/action/citation_signature","submit_replication":"https://pith.science/pith/Q2SVGR4YB7OY7KVP3NJ77FHDI2/action/replication_record"}},"created_at":"2026-07-05T09:15:42.929587+00:00","updated_at":"2026-07-05T09:15:42.929587+00:00"}