{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CMOVZH4P6WJWNFEDARSOEDCPVF","short_pith_number":"pith:CMOVZH4P","schema_version":"1.0","canonical_sha256":"131d5c9f8ff5936694830464e20c4fa94c0e65c93932716d62c60b894ed1dcbf","source":{"kind":"arxiv","id":"2412.09810","version":2},"attestation_state":"computed","paper":{"title":"The Complexity Dynamics of Grokking","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Branton DeMoss, Ingmar Posner, Jakob Foerster, Nick Hawes, Silvia Sapora","submitted_at":"2024-12-13T02:57:59Z","abstract_excerpt":"We demonstrate the existence of a complexity phase transition in neural networks by studying the grokking phenomenon, where networks suddenly transition from memorization to generalization long after overfitting their training data. To characterize this phase transition, we introduce a theoretical framework for measuring complexity based on rate-distortion theory and Kolmogorov complexity, which can be understood as principled lossy compression for networks. We find that properly regularized networks exhibit a sharp phase transition: complexity rises during memorization, then falls as the netw"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.09810","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-13T02:57:59Z","cross_cats_sorted":[],"title_canon_sha256":"0f850d0f16df7890732f6f8a3461dfafd730d82cc32f31e75a04647dab042384","abstract_canon_sha256":"4199a5440a55c9b29070e41673e47225d7fc4894a221213fa3ea4ad3babdb95d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:56:51.612614Z","signature_b64":"a3feYKJIubr7JpUPsneXtU2JbQIKlsbrGIVhAIDrRBZvKptd0xdZC995+Lk7WEZU06rH3nePfAhgN2AyZL0AAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"131d5c9f8ff5936694830464e20c4fa94c0e65c93932716d62c60b894ed1dcbf","last_reissued_at":"2026-07-05T11:56:51.612162Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:56:51.612162Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Complexity Dynamics of Grokking","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Branton DeMoss, Ingmar Posner, Jakob Foerster, Nick Hawes, Silvia Sapora","submitted_at":"2024-12-13T02:57:59Z","abstract_excerpt":"We demonstrate the existence of a complexity phase transition in neural networks by studying the grokking phenomenon, where networks suddenly transition from memorization to generalization long after overfitting their training data. To characterize this phase transition, we introduce a theoretical framework for measuring complexity based on rate-distortion theory and Kolmogorov complexity, which can be understood as principled lossy compression for networks. We find that properly regularized networks exhibit a sharp phase transition: complexity rises during memorization, then falls as the netw"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.09810","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.09810/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.09810","created_at":"2026-07-05T11:56:51.612219+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.09810v2","created_at":"2026-07-05T11:56:51.612219+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.09810","created_at":"2026-07-05T11:56:51.612219+00:00"},{"alias_kind":"pith_short_12","alias_value":"CMOVZH4P6WJW","created_at":"2026-07-05T11:56:51.612219+00:00"},{"alias_kind":"pith_short_16","alias_value":"CMOVZH4P6WJWNFED","created_at":"2026-07-05T11:56:51.612219+00:00"},{"alias_kind":"pith_short_8","alias_value":"CMOVZH4P","created_at":"2026-07-05T11:56:51.612219+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06639","citing_title":"At-Grok Is Not Converged:A Measurement-Validity Audit for Grokking Representation Metrics","ref_index":6,"is_internal_anchor":true},{"citing_arxiv_id":"2504.20571","citing_title":"Reinforcement Learning for Reasoning in Large Language Models with One Training Example","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09724","citing_title":"Model Capacity Determines Grokking through Competing Memorisation and Generalisation Speeds","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CMOVZH4P6WJWNFEDARSOEDCPVF","json":"https://pith.science/pith/CMOVZH4P6WJWNFEDARSOEDCPVF.json","graph_json":"https://pith.science/api/pith-number/CMOVZH4P6WJWNFEDARSOEDCPVF/graph.json","events_json":"https://pith.science/api/pith-number/CMOVZH4P6WJWNFEDARSOEDCPVF/events.json","paper":"https://pith.science/paper/CMOVZH4P"},"agent_actions":{"view_html":"https://pith.science/pith/CMOVZH4P6WJWNFEDARSOEDCPVF","download_json":"https://pith.science/pith/CMOVZH4P6WJWNFEDARSOEDCPVF.json","view_paper":"https://pith.science/paper/CMOVZH4P","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.09810&json=true","fetch_graph":"https://pith.science/api/pith-number/CMOVZH4P6WJWNFEDARSOEDCPVF/graph.json","fetch_events":"https://pith.science/api/pith-number/CMOVZH4P6WJWNFEDARSOEDCPVF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CMOVZH4P6WJWNFEDARSOEDCPVF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CMOVZH4P6WJWNFEDARSOEDCPVF/action/storage_attestation","attest_author":"https://pith.science/pith/CMOVZH4P6WJWNFEDARSOEDCPVF/action/author_attestation","sign_citation":"https://pith.science/pith/CMOVZH4P6WJWNFEDARSOEDCPVF/action/citation_signature","submit_replication":"https://pith.science/pith/CMOVZH4P6WJWNFEDARSOEDCPVF/action/replication_record"}},"created_at":"2026-07-05T11:56:51.612219+00:00","updated_at":"2026-07-05T11:56:51.612219+00:00"}