{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IVIGOX7UF235ULHMYIPZADXP72","short_pith_number":"pith:IVIGOX7U","schema_version":"1.0","canonical_sha256":"4550675ff42eb7da2cecc21f900eeffebbaa8e115bed55aeff902f6296a5e132","source":{"kind":"arxiv","id":"2404.07129","version":1},"attestation_state":"computed","paper":{"title":"What needs to go right for an induction head? A mechanistic study of in-context learning circuits and their formation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aaditya K. Singh, Andrew M. Saxe, Felix Hill, Stephanie C.Y. Chan, Ted Moskovitz","submitted_at":"2024-04-10T16:07:38Z","abstract_excerpt":"In-context learning is a powerful emergent ability in transformer models. Prior work in mechanistic interpretability has identified a circuit element that may be critical for in-context learning -- the induction head (IH), which performs a match-and-copy operation. During training of large transformers on natural language data, IHs emerge around the same time as a notable phase change in the loss. Despite the robust evidence for IHs and this interesting coincidence with the phase change, relatively little is known about the diversity and emergence dynamics of IHs. Why is there more than one IH"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.07129","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-04-10T16:07:38Z","cross_cats_sorted":[],"title_canon_sha256":"0a9817c368d85ddfa6e01311df744232143c6269a92ca6a38715f96e51482cfb","abstract_canon_sha256":"b465abfbf8631e602d8a597defa32866c42475d8f9937c786e67d6a07dd37f2b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:06:35.636540Z","signature_b64":"my1TEMOuSvigE4BDK0QzrTiQARhXVbUNdsxSakgfNlPU90z7pon7cog564XBpajD5QuyiALkeR5mhHUw0CrvAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4550675ff42eb7da2cecc21f900eeffebbaa8e115bed55aeff902f6296a5e132","last_reissued_at":"2026-07-05T08:06:35.635955Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:06:35.635955Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"What needs to go right for an induction head? A mechanistic study of in-context learning circuits and their formation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aaditya K. Singh, Andrew M. Saxe, Felix Hill, Stephanie C.Y. Chan, Ted Moskovitz","submitted_at":"2024-04-10T16:07:38Z","abstract_excerpt":"In-context learning is a powerful emergent ability in transformer models. Prior work in mechanistic interpretability has identified a circuit element that may be critical for in-context learning -- the induction head (IH), which performs a match-and-copy operation. During training of large transformers on natural language data, IHs emerge around the same time as a notable phase change in the loss. Despite the robust evidence for IHs and this interesting coincidence with the phase change, relatively little is known about the diversity and emergence dynamics of IHs. Why is there more than one IH"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.07129","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.07129/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.07129","created_at":"2026-07-05T08:06:35.636022+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.07129v1","created_at":"2026-07-05T08:06:35.636022+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.07129","created_at":"2026-07-05T08:06:35.636022+00:00"},{"alias_kind":"pith_short_12","alias_value":"IVIGOX7UF235","created_at":"2026-07-05T08:06:35.636022+00:00"},{"alias_kind":"pith_short_16","alias_value":"IVIGOX7UF235ULHM","created_at":"2026-07-05T08:06:35.636022+00:00"},{"alias_kind":"pith_short_8","alias_value":"IVIGOX7U","created_at":"2026-07-05T08:06:35.636022+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.06621","citing_title":"Fingerprint, Not Blueprint: How Positional Schemes Set the Default Spectral Algebra of Attention","ref_index":16,"is_internal_anchor":true},{"citing_arxiv_id":"2607.07316","citing_title":"Mechanistic Interpretability for Neural Networks: Circuits, Sparse Features and Symbolic Reasoning","ref_index":15,"is_internal_anchor":true},{"citing_arxiv_id":"2506.04289","citing_title":"Relational reasoning and inductive bias in transformers and large language models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2507.20906","citing_title":"Soft Head Selection for Injecting ICL-Derived Task Embeddings","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2509.24164","citing_title":"Localizing Task Recognition and Task Learning in In-Context Learning via Attention Head Analysis","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08405","citing_title":"Belief or Circuitry? Causal Evidence for In-Context Graph Learning","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13403","citing_title":"Why Multimodal In-Context Learning Lags Behind? Unveiling the Inner Mechanisms and Bottlenecks","ref_index":37,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IVIGOX7UF235ULHMYIPZADXP72","json":"https://pith.science/pith/IVIGOX7UF235ULHMYIPZADXP72.json","graph_json":"https://pith.science/api/pith-number/IVIGOX7UF235ULHMYIPZADXP72/graph.json","events_json":"https://pith.science/api/pith-number/IVIGOX7UF235ULHMYIPZADXP72/events.json","paper":"https://pith.science/paper/IVIGOX7U"},"agent_actions":{"view_html":"https://pith.science/pith/IVIGOX7UF235ULHMYIPZADXP72","download_json":"https://pith.science/pith/IVIGOX7UF235ULHMYIPZADXP72.json","view_paper":"https://pith.science/paper/IVIGOX7U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.07129&json=true","fetch_graph":"https://pith.science/api/pith-number/IVIGOX7UF235ULHMYIPZADXP72/graph.json","fetch_events":"https://pith.science/api/pith-number/IVIGOX7UF235ULHMYIPZADXP72/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IVIGOX7UF235ULHMYIPZADXP72/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IVIGOX7UF235ULHMYIPZADXP72/action/storage_attestation","attest_author":"https://pith.science/pith/IVIGOX7UF235ULHMYIPZADXP72/action/author_attestation","sign_citation":"https://pith.science/pith/IVIGOX7UF235ULHMYIPZADXP72/action/citation_signature","submit_replication":"https://pith.science/pith/IVIGOX7UF235ULHMYIPZADXP72/action/replication_record"}},"created_at":"2026-07-05T08:06:35.636022+00:00","updated_at":"2026-07-05T08:06:35.636022+00:00"}