{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BYZQCFHXFKH6XA3ICRMRI4DHM2","short_pith_number":"pith:BYZQCFHX","schema_version":"1.0","canonical_sha256":"0e330114f72a8feb8368145914706766a548cc5dc1793b4b5c7b5a03fd8490a3","source":{"kind":"arxiv","id":"2409.19150","version":1},"attestation_state":"computed","paper":{"title":"On the Power of Decision Trees in Auto-Regressive Language Modeling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Eran Malach, Tomaso Poggio, Tomer Galanti, Yulu Gan","submitted_at":"2024-09-27T21:25:00Z","abstract_excerpt":"Originally proposed for handling time series data, Auto-regressive Decision Trees (ARDTs) have not yet been explored for language modeling. This paper delves into both the theoretical and practical applications of ARDTs in this new context. We theoretically demonstrate that ARDTs can compute complex functions, such as simulating automata, Turing machines, and sparse circuits, by leveraging \"chain-of-thought\" computations. Our analysis provides bounds on the size, depth, and computational efficiency of ARDTs, highlighting their surprising computational power. Empirically, we train ARDTs on simp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.19150","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-09-27T21:25:00Z","cross_cats_sorted":[],"title_canon_sha256":"04e949b97e2f698f77d0c93b568f6e1e2542cbacf04e4a0b9442ad97fbf22e14","abstract_canon_sha256":"1a3be6147220253f1b86e75645162bcbe827454950611f316f40f180de360313"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:13:11.311457Z","signature_b64":"lf8wNfQee/ehlpHM8FRtyzhLTUNsDcOeGfcAcaWD6vZ7HW9TffmILXYgJLg0OPZHQIMygc5OmGklxE3ZySk2Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0e330114f72a8feb8368145914706766a548cc5dc1793b4b5c7b5a03fd8490a3","last_reissued_at":"2026-07-05T09:13:11.311049Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:13:11.311049Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the Power of Decision Trees in Auto-Regressive Language Modeling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Eran Malach, Tomaso Poggio, Tomer Galanti, Yulu Gan","submitted_at":"2024-09-27T21:25:00Z","abstract_excerpt":"Originally proposed for handling time series data, Auto-regressive Decision Trees (ARDTs) have not yet been explored for language modeling. This paper delves into both the theoretical and practical applications of ARDTs in this new context. We theoretically demonstrate that ARDTs can compute complex functions, such as simulating automata, Turing machines, and sparse circuits, by leveraging \"chain-of-thought\" computations. Our analysis provides bounds on the size, depth, and computational efficiency of ARDTs, highlighting their surprising computational power. Empirically, we train ARDTs on simp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.19150","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.19150/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.19150","created_at":"2026-07-05T09:13:11.311106+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.19150v1","created_at":"2026-07-05T09:13:11.311106+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.19150","created_at":"2026-07-05T09:13:11.311106+00:00"},{"alias_kind":"pith_short_12","alias_value":"BYZQCFHXFKH6","created_at":"2026-07-05T09:13:11.311106+00:00"},{"alias_kind":"pith_short_16","alias_value":"BYZQCFHXFKH6XA3I","created_at":"2026-07-05T09:13:11.311106+00:00"},{"alias_kind":"pith_short_8","alias_value":"BYZQCFHX","created_at":"2026-07-05T09:13:11.311106+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.02550","citing_title":"Position: A Theory of Deep Learning Must Include Compositional Sparsity","ref_index":25,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BYZQCFHXFKH6XA3ICRMRI4DHM2","json":"https://pith.science/pith/BYZQCFHXFKH6XA3ICRMRI4DHM2.json","graph_json":"https://pith.science/api/pith-number/BYZQCFHXFKH6XA3ICRMRI4DHM2/graph.json","events_json":"https://pith.science/api/pith-number/BYZQCFHXFKH6XA3ICRMRI4DHM2/events.json","paper":"https://pith.science/paper/BYZQCFHX"},"agent_actions":{"view_html":"https://pith.science/pith/BYZQCFHXFKH6XA3ICRMRI4DHM2","download_json":"https://pith.science/pith/BYZQCFHXFKH6XA3ICRMRI4DHM2.json","view_paper":"https://pith.science/paper/BYZQCFHX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.19150&json=true","fetch_graph":"https://pith.science/api/pith-number/BYZQCFHXFKH6XA3ICRMRI4DHM2/graph.json","fetch_events":"https://pith.science/api/pith-number/BYZQCFHXFKH6XA3ICRMRI4DHM2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BYZQCFHXFKH6XA3ICRMRI4DHM2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BYZQCFHXFKH6XA3ICRMRI4DHM2/action/storage_attestation","attest_author":"https://pith.science/pith/BYZQCFHXFKH6XA3ICRMRI4DHM2/action/author_attestation","sign_citation":"https://pith.science/pith/BYZQCFHXFKH6XA3ICRMRI4DHM2/action/citation_signature","submit_replication":"https://pith.science/pith/BYZQCFHXFKH6XA3ICRMRI4DHM2/action/replication_record"}},"created_at":"2026-07-05T09:13:11.311106+00:00","updated_at":"2026-07-05T09:13:11.311106+00:00"}