{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:J7WOKSPANUSS5VHWN6AM5P66Z7","short_pith_number":"pith:J7WOKSPA","schema_version":"1.0","canonical_sha256":"4fece549e06d252ed4f66f80cebfdecff7f16d67b83945998b5a4ead70c11fc1","source":{"kind":"arxiv","id":"2303.11873","version":1},"attestation_state":"computed","paper":{"title":"A Tale of Two Circuits: Grokking as Competition of Sparse and Dense Subnetworks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aman Shukla, Nikolaos Tsilivis, William Merrill","submitted_at":"2023-03-21T14:17:29Z","abstract_excerpt":"Grokking is a phenomenon where a model trained on an algorithmic task first overfits but, then, after a large amount of additional training, undergoes a phase transition to generalize perfectly. We empirically study the internal structure of networks undergoing grokking on the sparse parity task, and find that the grokking phase transition corresponds to the emergence of a sparse subnetwork that dominates model predictions. On an optimization level, we find that this subnetwork arises when a small subset of neurons undergoes rapid norm growth, whereas the other neurons in the network decay slo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.11873","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-03-21T14:17:29Z","cross_cats_sorted":[],"title_canon_sha256":"03dc2f9e61afcb919b45d185bd02c0aa24ff4f19b932543b1e226d2f095c68dd","abstract_canon_sha256":"f82caaf97da75923fadf8e57bbb56e959671709649435cc0f726344a9924f3b0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:53:23.198544Z","signature_b64":"Uiq+Zd8svTgDl9VA1YC1AzZp4U/r4vZG4K8XpCok1hgqFVEamHOQd327nIInf0Z+lXVaUu7+1DGewfRHVvE5CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4fece549e06d252ed4f66f80cebfdecff7f16d67b83945998b5a4ead70c11fc1","last_reissued_at":"2026-07-05T05:53:23.198083Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:53:23.198083Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Tale of Two Circuits: Grokking as Competition of Sparse and Dense Subnetworks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aman Shukla, Nikolaos Tsilivis, William Merrill","submitted_at":"2023-03-21T14:17:29Z","abstract_excerpt":"Grokking is a phenomenon where a model trained on an algorithmic task first overfits but, then, after a large amount of additional training, undergoes a phase transition to generalize perfectly. We empirically study the internal structure of networks undergoing grokking on the sparse parity task, and find that the grokking phase transition corresponds to the emergence of a sparse subnetwork that dominates model predictions. On an optimization level, we find that this subnetwork arises when a small subset of neurons undergoes rapid norm growth, whereas the other neurons in the network decay slo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.11873","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.11873/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.11873","created_at":"2026-07-05T05:53:23.198139+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.11873v1","created_at":"2026-07-05T05:53:23.198139+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.11873","created_at":"2026-07-05T05:53:23.198139+00:00"},{"alias_kind":"pith_short_12","alias_value":"J7WOKSPANUSS","created_at":"2026-07-05T05:53:23.198139+00:00"},{"alias_kind":"pith_short_16","alias_value":"J7WOKSPANUSS5VHW","created_at":"2026-07-05T05:53:23.198139+00:00"},{"alias_kind":"pith_short_8","alias_value":"J7WOKSPA","created_at":"2026-07-05T05:53:23.198139+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08350","citing_title":"Grokking and epoch-wise double descent in quantum neural networks","ref_index":3,"is_internal_anchor":true},{"citing_arxiv_id":"2606.22873","citing_title":"SingGuard: A Policy-Adaptive Multimodal LLM Guardrail with Dynamic Reasoning","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12966","citing_title":"Circuit Synchronization Precedes Generalization: A Causal Precursor to Grokking","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13125","citing_title":"Select and Improve: Understanding the Mechanics of Post-Training for Reasoning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22873","citing_title":"SingGuard: A Policy-Adaptive Multimodal LLM Guardrail with Dynamic Reasoning","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2606.32000","citing_title":"Radial Suppression Accelerates Algorithmic Generalization: A Geometric Analysis of Delayed Generalization","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04930","citing_title":"Egalitarian Gradient Descent: A Simple Approach to Accelerated Grokking","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2602.16746","citing_title":"Low-Dimensional and Transversely Curved Optimization Dynamics in Grokking","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2602.18523","citing_title":"The Geometry of Multi-Task Grokking: Transverse Instability, Superposition, and Weight Decay Phase Structure","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13123","citing_title":"Spectral Entropy Collapse as a Phase Transition in Delayed Generalisation: An Interventional and Predictive Framework for Grokkin","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09724","citing_title":"Model Capacity Determines Grokking through Competing Memorisation and Generalisation Speeds","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05683","citing_title":"Spectral Lens: Activation and Gradient Spectra as Diagnostics of LLM Optimization","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13123","citing_title":"Spectral Entropy Collapse as a Phase Transition in Delayed Generalisation: An Interventional and Predictive Framework for Grokkin","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J7WOKSPANUSS5VHWN6AM5P66Z7","json":"https://pith.science/pith/J7WOKSPANUSS5VHWN6AM5P66Z7.json","graph_json":"https://pith.science/api/pith-number/J7WOKSPANUSS5VHWN6AM5P66Z7/graph.json","events_json":"https://pith.science/api/pith-number/J7WOKSPANUSS5VHWN6AM5P66Z7/events.json","paper":"https://pith.science/paper/J7WOKSPA"},"agent_actions":{"view_html":"https://pith.science/pith/J7WOKSPANUSS5VHWN6AM5P66Z7","download_json":"https://pith.science/pith/J7WOKSPANUSS5VHWN6AM5P66Z7.json","view_paper":"https://pith.science/paper/J7WOKSPA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.11873&json=true","fetch_graph":"https://pith.science/api/pith-number/J7WOKSPANUSS5VHWN6AM5P66Z7/graph.json","fetch_events":"https://pith.science/api/pith-number/J7WOKSPANUSS5VHWN6AM5P66Z7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J7WOKSPANUSS5VHWN6AM5P66Z7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J7WOKSPANUSS5VHWN6AM5P66Z7/action/storage_attestation","attest_author":"https://pith.science/pith/J7WOKSPANUSS5VHWN6AM5P66Z7/action/author_attestation","sign_citation":"https://pith.science/pith/J7WOKSPANUSS5VHWN6AM5P66Z7/action/citation_signature","submit_replication":"https://pith.science/pith/J7WOKSPANUSS5VHWN6AM5P66Z7/action/replication_record"}},"created_at":"2026-07-05T05:53:23.198139+00:00","updated_at":"2026-07-05T05:53:23.198139+00:00"}