{"as_of":"2026-08-15T02:38:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:46fe2618ecea6b1c81e6846ee09dd9368d04686a90c74d7d03b5b384660ff4f4","coverage":[{"denominator":53,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":53,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T19:31:23.888184Z","state":"measured"},{"denominator":54,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":54,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-29T22:53:30.561450Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-06-29T22:54:00.682463Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"cited_work":{"arxiv_id":"2507.05644","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.05644","snapshot_observed_at":"2026-06-29T22:54:00.682463Z","title":"Boix-Adsera, N","venue":null,"work_id":"5b61d926-bb01-4536-adf4-04c6e8862bff","year":2026},"citing_paper":{"arxiv_id":"2606.07563","last_updated":"2026-05-25T18:32:52Z","snapshot_observed_at":"2026-08-10T22:55:29.834508Z","submitted_at":"2026-05-25T18:32:52Z","title":"Emergence via Phase Transitions: Mechanism Landscapes and Universal Convergence Across Complex Systems","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-29T22:53:30.561450Z"},"links":{"cited_paper":"/paper/2507.05644","citing_paper":"/paper/2606.07563"},"observation_digest":"sha256:a847bac1fc50b8a79cd2dcc543a596f6daf7079fa861601a163cb34022c849e8","observation_id":"ff9315d3-5ca8-48f3-8eec-64cd09cffed3","resolution":{"observed_at":"2026-06-29T22:54:00.683913Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.05644/citation-record","integrity":"/paper/2507.05644/integrity","json":"/paper/2507.05644/citation-record.json","paper":"/paper/2507.05644"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:25.033265Z","title":null,"venue":null,"work_id":"84420f34-53ec-4f20-a43f-478a68f9cad6","year":2022},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:18.105519Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:e7dd5027fffeaf36e0c6b0514e47a41879f3e9372552f6a4168cb2f154e02962","observation_id":"d2a7dd51-c5bb-4e64-ba91-6e51baf603be","resolution":{"observed_at":"2026-08-06T19:31:25.037471Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:25.019473Z","title":null,"venue":null,"work_id":"dd41bd11-337e-448d-a12d-7ab11d6298d3","year":2023},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:18.205740Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:5c8eed45460b9e25c35276ac7b1704b1c0a122b2a2be8fe051d2f40c237686b4","observation_id":"8dc1e045-a237-4cdb-9d54-d6c85e0aae40","resolution":{"observed_at":"2026-08-06T19:31:25.023591Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:25.005712Z","title":"Arora, N","venue":null,"work_id":"f31417e3-069e-4f59-a472-0fae38cae75b","year":2019},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:18.338909Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:4c63b5bac52601040e75b25cce91e6eebcb6f9dd7bf82e0a9ac3749f0f30fe94","observation_id":"1c2029c5-56d5-46d8-b07b-08b1e357365f","resolution":{"observed_at":"2026-08-06T19:31:25.009845Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.991963Z","title":"Arora, N","venue":null,"work_id":"8efa5d94-0487-4b6e-9e22-174aceba9658","year":2018},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:18.437898Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:0d58a3e7a185c9e7254bb045d317e42e4aca7f9278626602807e8362b80eeb77","observation_id":"7dc1aa41-710e-4d04-9812-e03eea3f77be","resolution":{"observed_at":"2026-08-06T19:31:24.996016Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.977939Z","title":"Arora, N","venue":null,"work_id":"76422e8b-7bf7-4c38-bc92-41f825032c50","year":2019},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:18.539500Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:1cba17abc79a00f376d6b9634e55253c38632b7013addb8701431b4aa3cafcfd","observation_id":"aefc0213-27b4-476a-8c64-43c8cf7fb730","resolution":{"observed_at":"2026-08-06T19:31:24.982231Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.964522Z","title":null,"venue":null,"work_id":"22085176-33fb-483f-a423-c3b63b8e1710","year":2021},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:18.640051Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:9400a16e63c516af36d867f743341660a32e958a2d5c247542ba0174f3db4baa","observation_id":"6e12cb4f-73fb-49c0-9a85-0223646c5a95","resolution":{"observed_at":"2026-08-06T19:31:24.968565Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.950424Z","title":"Barak, B","venue":null,"work_id":"e1a5d074-1b8c-4176-a0cb-58b55f14ce51","year":2022},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:18.786168Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:3c14130e9c0b2adad4f742dd19e0cdf8c77aa1ff73146bd965290cc2ef0d7f70","observation_id":"fc3c05ed-b535-4bb1-bc4e-601881048331","resolution":{"observed_at":"2026-08-06T19:31:24.955034Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.03708","last_updated":"2025-05-28T19:52:55Z","snapshot_observed_at":"2026-08-14T19:57:35.412572Z","submitted_at":"2025-02-06T01:41:48Z","title":"Toward universal steering and monitoring of AI models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.03708","snapshot_observed_at":"2026-08-06T19:31:18.912515Z","title":"Beaglehole, A","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:18.912515Z"},"links":{"cited_paper":"/paper/2502.03708","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:6f70767c22afbbeac8e79cb1a8a4c40ac39536824fd0c1faa4a7c4a5c616529c","observation_id":"72763c8e-e5cf-4e8f-9e15-48f13c44e7d0","resolution":{"observed_at":"2026-08-06T19:31:18.912515Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.00570","last_updated":"2023-09-01T16:30:02Z","snapshot_observed_at":"2026-08-13T10:22:30.530873Z","submitted_at":"2023-09-01T16:30:02Z","title":"Mechanism of feature learning in convolutional neural networks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.00570","snapshot_observed_at":"2026-08-06T19:31:19.028549Z","title":"Beaglehole, A","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:19.028549Z"},"links":{"cited_paper":"/paper/2309.00570","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:f96263adcf9fdc1bdeb0d64bde921dc7dc41a64ba75f3d8602355b8cdd25e841","observation_id":"88e5ce86-8a4e-427b-ad8d-028614980bb1","resolution":{"observed_at":"2026-08-06T19:31:19.028549Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02984","last_updated":"2024-02-20T19:17:52Z","snapshot_observed_at":"2026-08-14T06:20:36.711581Z","submitted_at":"2023-10-04T17:20:34Z","title":"Scaling Laws for Associative Memories","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02984","snapshot_observed_at":"2026-08-06T19:31:19.171471Z","title":"Cabannes, E","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:19.171471Z"},"links":{"cited_paper":"/paper/2310.02984","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:bffc1bc4543ed7ce2fdca939893cdd218fafabe7fa3a2ac192740cbc183e8a3f","observation_id":"b81b91af-df0c-4091-a82e-b4c499256fed","resolution":{"observed_at":"2026-08-06T19:31:19.171471Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18724","last_updated":"2024-02-28T21:47:30Z","snapshot_observed_at":"2026-08-14T06:20:52.578063Z","submitted_at":"2024-02-28T21:47:30Z","title":"Learning Associative Memories with Gradient Descent","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.18724","snapshot_observed_at":"2026-08-06T19:31:19.335976Z","title":"Cabannes, B","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:19.335976Z"},"links":{"cited_paper":"/paper/2402.18724","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:59fb700386fe6b61175255ef4fcef0ce4aec58aabee2622f0b99a8e89ac79300","observation_id":"94ed4ca1-0948-41bf-b368-c8583e1764af","resolution":{"observed_at":"2026-08-06T19:31:19.335976Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.936391Z","title":"Damian, J","venue":null,"work_id":"4d3b54fa-c487-46be-95dd-cc37cadfbbe6","year":2022},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:19.459944Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:bd78a598365324c5d520e68e38105e71a749857446574b2c3f13baca9ec19543","observation_id":"364be1d2-467b-4b3d-ba35-6b2fd9234541","resolution":{"observed_at":"2026-08-06T19:31:24.940650Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.921714Z","title":"Davis and W","venue":null,"work_id":"b95ec9f8-636e-4c10-b24b-75191d6ed88f","year":1970},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:19.606152Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:f3bc8f83ba00fe9506e15b6c82afdeed4ac67f4bdc226abc74f82549b54039ff","observation_id":"a08c7306-3f91-4f51-9c8d-7d5767e9cbf9","resolution":{"observed_at":"2026-08-06T19:31:24.926141Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.11004","last_updated":"2024-02-16T18:28:36Z","snapshot_observed_at":"2026-08-14T16:39:29.734755Z","submitted_at":"2024-02-16T18:28:36Z","title":"The Evolution of Statistical Induction Heads: In-Context Learning Markov Chains","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.11004","snapshot_observed_at":"2026-08-06T19:31:19.725776Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:19.725776Z"},"links":{"cited_paper":"/paper/2402.11004","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:f53f00871c7a5d27ba7eaf7dcc87a5da8539e615a7ab9b3bfc25d9fbaa1cf21b","observation_id":"ddb929c6-eeb2-4083-b98d-6709255ca34d","resolution":{"observed_at":"2026-08-06T19:31:19.725776Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.906805Z","title":null,"venue":null,"work_id":"93d61bbc-8cf3-49ed-b55f-07f5bc01ca09","year":2023},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:19.908449Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:a2b6a5860a7e270c35316fe149dfc74dee3e73b3a4001afd8d127b71c6c68b8c","observation_id":"2eea3257-3082-4f41-bcb0-bd63cafb9b10","resolution":{"observed_at":"2026-08-06T19:31:24.911625Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.892962Z","title":"Fernandez-Delgado, E","venue":null,"work_id":"28470f86-c177-4f60-aa55-eb41c17d0869","year":2014},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:20.081513Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:8042451b8d0247e46061f04f76d9d6132d93ef1710897dc8fcbe2d8a2cb37792","observation_id":"c4371282-5327-4e9f-90cc-ff919e938528","resolution":{"observed_at":"2026-08-06T19:31:24.897099Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.878958Z","title":null,"venue":null,"work_id":"883af1ce-36af-455a-a702-30c51a2ca855","year":2024},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:20.153867Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:8f53ba388723a5a13901d646b791683241339a6aea3b71dc9baa59508a8d23da","observation_id":"4c035b1c-aeb2-432e-bd61-0f68a23ebe64","resolution":{"observed_at":"2026-08-06T19:31:24.883272Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.05794","last_updated":"2024-10-18T21:32:39Z","snapshot_observed_at":"2026-08-13T15:25:18.045566Z","submitted_at":"2022-06-12T17:06:35Z","title":"SGD and Weight Decay Secretly Minimize the Rank of Your Neural Network","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.05794","snapshot_observed_at":"2026-08-06T19:31:20.239832Z","title":"Galanti, Z","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:20.239832Z"},"links":{"cited_paper":"/paper/2206.05794","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:60706ba05d3c582d469ddb43bb6e87a3b126d286312a1a9206a712bfc9d0e760","observation_id":"706d7da5-4244-4ea5-aa3d-85df53b06cf3","resolution":{"observed_at":"2026-08-06T19:31:20.239832Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.862974Z","title":"Gan and T","venue":null,"work_id":"020aa732-ca99-4a0f-9aa1-aafe39875b8f","year":2024},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:20.334218Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:5f6c6c10c6c96dd06a22d0304654a68bc9ffc69dd6e81f5e13d54384ed8a31c4","observation_id":"8f85cd05-c950-4fb7-8c4b-4d2df31c04a4","resolution":{"observed_at":"2026-08-06T19:31:24.867454Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.02679","last_updated":"2023-01-06T19:00:01Z","snapshot_observed_at":"2026-08-13T13:08:38.924753Z","submitted_at":"2023-01-06T19:00:01Z","title":"Grokking modular arithmetic","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.02679","snapshot_observed_at":"2026-08-06T19:31:20.404668Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:20.404668Z"},"links":{"cited_paper":"/paper/2301.02679","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:a48eb66eef19761b6d70226649ce4d338911774fecb95fd70969d5e78f3556e9","observation_id":"8b77646c-ee6e-4b99-89e7-2011278f6431","resolution":{"observed_at":"2026-08-06T19:31:20.404668Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.847750Z","title":"Gunasekar, J","venue":null,"work_id":"f9d6c54e-1c98-4406-aad3-ab4799950b90","year":2018},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:20.464036Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:0e54d4d6554f9667da8bd7a54c63ac3795d24a74b80393b109947427cf6b0ba5","observation_id":"aa502f56-32bf-400a-b429-ac33cba44b8b","resolution":{"observed_at":"2026-08-06T19:31:24.852195Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.833629Z","title":"Gunasekar, B","venue":null,"work_id":"730ba1df-162b-4c31-89e3-b729a70ac6bd","year":2017},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:20.524900Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:503f1ced7ae27c831f0ab109373997b188239ad9bff91f2f2b9630d3839a6bcc","observation_id":"cd7e5fd8-d43d-4135-8fbc-21f33ddc52a2","resolution":{"observed_at":"2026-08-06T19:31:24.837780Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.02073","last_updated":"2022-05-10T01:55:43Z","snapshot_observed_at":"2026-07-06T11:15:46.517682Z","submitted_at":"2021-06-03T18:31:41Z","title":"Neural Collapse Under MSE Loss: Proximity to and Dynamics on the Central Path","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.02073","snapshot_observed_at":"2026-08-06T19:31:20.601467Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:20.601467Z"},"links":{"cited_paper":"/paper/2106.02073","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:5b2e9490b7a4d1e41186bc1b5ff14d76785ef85ff6bb98344350be27716abcad","observation_id":"1f07f671-682f-432a-9c04-4f42f1180631","resolution":{"observed_at":"2026-08-06T19:31:20.601467Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.819469Z","title":"Jacot, F","venue":null,"work_id":"6fb19b1b-8b5d-4cf1-9738-19c29e82c3f4","year":2018},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:20.687675Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:43c62c7e976ce0ec8394f11d4c4da83d2cf217d65523b32abbc4a2fc7b579059","observation_id":"b978ebb9-e0bd-47df-a37d-d1cf202c3168","resolution":{"observed_at":"2026-08-06T19:31:24.823514Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.804511Z","title":"Ji and M","venue":null,"work_id":"5bc42675-bb72-48e4-aeda-d39cea43262b","year":2019},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:20.774219Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:6170b30fc8a4b36f688bac6f2d9ccb9a6125f7a90c79a976344ba80e2614e515","observation_id":"908d1050-2f3c-4e82-8e92-abb7418a68c3","resolution":{"observed_at":"2026-08-06T19:31:24.808800Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.789302Z","title":"Ji and M","venue":null,"work_id":"ea92526e-3e95-4249-a60c-5868c4cf3386","year":2020},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:20.899292Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:b4561f14c64b0cde3b2b3fe8806d93b5425f4f5b5086949a06f7eda5b4865205","observation_id":"3f861106-d511-430a-8734-bee57e9b49d0","resolution":{"observed_at":"2026-08-06T19:31:24.794164Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.04041","last_updated":"2023-04-11T06:11:14Z","snapshot_observed_at":"2026-08-13T15:27:51.024747Z","submitted_at":"2022-06-08T17:55:28Z","title":"Neural Collapse: A Review on Modelling Principles and Generalization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.04041","snapshot_observed_at":"2026-08-06T19:31:20.976793Z","title":"Kothapalli","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:20.976793Z"},"links":{"cited_paper":"/paper/2206.04041","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:4b4bcb188062dfeb03793814e2fc5b2ae5f0cd99e8806b124eacd69ed913ac92","observation_id":"950a49b5-cdf4-445e-8e06-ded169d96f66","resolution":{"observed_at":"2026-08-06T19:31:20.976793Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:21.064475Z","title":"Krizhevsky, G","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:21.064475Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:2c8984e2e60f46970b102f8a98aad37723562e574ffc6b0aaa9cb7cd4a0b68d2","observation_id":"c16ca467-e6f9-411b-9775-550facc9b7a1","resolution":{"observed_at":"2026-08-06T19:31:21.064475Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06110","last_updated":"2024-04-11T16:15:34Z","snapshot_observed_at":"2026-08-13T16:39:28.450010Z","submitted_at":"2023-10-09T19:33:21Z","title":"Grokking as the Transition from Lazy to Rich Training Dynamics","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06110","snapshot_observed_at":"2026-08-06T19:31:21.129096Z","title":"Kumar, B","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:21.129096Z"},"links":{"cited_paper":"/paper/2310.06110","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:ad9f94fd4ff47a463119f579abce815411aa4231907a09116883fb65fb4ab719","observation_id":"59989079-aa73-4069-bec2-1dbc2a1823e9","resolution":{"observed_at":"2026-08-06T19:31:21.129096Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.763798Z","title":null,"venue":null,"work_id":"8f2ac8c9-2909-46a0-b40d-93d95ff5f643","year":1998},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:21.211221Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:cb371c41b48f849876e8709c790071c7b1ce5b348c7e87b6dba9ba157d93bc4d","observation_id":"d29c9979-b953-4e38-a1bf-c955c02af362","resolution":{"observed_at":"2026-08-06T19:31:24.768034Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2105.10585","last_updated":"2021-12-13T17:43:52Z","snapshot_observed_at":"2026-07-06T11:11:45.392079Z","submitted_at":"2021-05-21T21:50:18Z","title":"Properties of the After Kernel","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2105.10585","snapshot_observed_at":"2026-08-06T19:31:21.269704Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:21.269704Z"},"links":{"cited_paper":"/paper/2105.10585","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:4cccc5533132e2c7cb54919b9af1624f0e2524284a6d8c8e8b6ae4a57e7764f3","observation_id":"c5225fb2-4470-4553-b909-8b34d624a417","resolution":{"observed_at":"2026-08-06T19:31:21.269704Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1906.05890","last_updated":"2020-12-29T05:33:37Z","snapshot_observed_at":"2026-08-15T01:17:45.378410Z","submitted_at":"2019-06-13T18:52:00Z","title":"Gradient Descent Maximizes the Margin of Homogeneous Neural Networks","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1906.05890","snapshot_observed_at":"2026-08-06T19:31:21.355070Z","title":"Lyu and J","venue":null,"work_id":null,"year":1906},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:21.355070Z"},"links":{"cited_paper":"/paper/1906.05890","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:6290f05f0ed01d7d9327bd69e46a5c30746b8bbdfd9c1c70831f7c2486273cc3","observation_id":"2230e2cb-89dc-40ef-a786-ec6da92d6e98","resolution":{"observed_at":"2026-08-06T19:31:21.355070Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.748310Z","title":"Mallinar, D","venue":null,"work_id":"47d40f84-a94c-4cdc-a539-4db03a8c5040","year":2025},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:21.505786Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:57ae32c3c65dbf76fad4060fdd5dd46aa4a709ae64b57d76813d588cc1b85d4e","observation_id":"ec1e8e2a-d381-4382-90bc-cbca99002a97","resolution":{"observed_at":"2026-08-06T19:31:24.752906Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.733236Z","title":"Marion and L","venue":null,"work_id":"7d3b001a-c105-4507-9141-2d5c31fec9d7","year":2024},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:21.564272Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:973481b8efb1a507f95eefd3982bc833d17748cb76ab119c5292018ba34249f3","observation_id":"e2ea96ef-6db6-42e7-a4ec-6f12e2b8c96e","resolution":{"observed_at":"2026-08-06T19:31:24.738055Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.718778Z","title":null,"venue":null,"work_id":"a80e9549-4cf0-4c5c-89c4-60eb7b477187","year":2023},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:21.649908Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:c268b252f82c0b5f11b05a48d249b476fd845a730fa8483bcd216fa254d29e48","observation_id":"731e296b-ac57-4d00-8e8d-241a89762000","resolution":{"observed_at":"2026-08-06T19:31:24.722954Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07568","last_updated":"2024-02-19T17:59:29Z","snapshot_observed_at":"2026-08-13T05:26:50.199513Z","submitted_at":"2023-11-13T18:56:33Z","title":"Feature emergence via margin maximization: case studies in algebraic tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07568","snapshot_observed_at":"2026-08-06T19:31:21.714527Z","title":"Morwani, B","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:21.714527Z"},"links":{"cited_paper":"/paper/2311.07568","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:498c93cc4dec47486124e71003597e4b13d831fdf7aeb492300a45ffd14becac","observation_id":"b25d254f-e221-4b1f-9701-47174cb8b4a0","resolution":{"observed_at":"2026-08-06T19:31:21.714527Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.05217","last_updated":"2023-10-19T21:25:32Z","snapshot_observed_at":"2026-08-02T10:20:00.635719Z","submitted_at":"2023-01-12T18:56:49Z","title":"Progress measures for grokking via mechanistic interpretability","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.05217","snapshot_observed_at":"2026-08-06T19:31:21.789558Z","title":"Nanda, L","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:21.789558Z"},"links":{"cited_paper":"/paper/2301.05217","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:0a7b60bfe2d0d4bfeab8cde5cadcdadecb591ace7d1e6eb8fd27d5bac14c7109","observation_id":"a4cb04f7-868d-4719-8839-5557a2c008b6","resolution":{"observed_at":"2026-08-06T19:31:21.789558Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14735","last_updated":"2024-08-13T15:45:37Z","snapshot_observed_at":"2026-08-14T21:12:01.094729Z","submitted_at":"2024-02-22T17:47:03Z","title":"How Transformers Learn Causal Structure with Gradient Descent","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14735","snapshot_observed_at":"2026-08-06T19:31:21.883499Z","title":"Nichani, A","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:21.883499Z"},"links":{"cited_paper":"/paper/2402.14735","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:a98a9c6dd93dfa8b58f449f298dad73c5647a71dd22701123642288fd3f51e5d","observation_id":"e6ad2f75-2630-4327-84e0-ed74ac02d5f6","resolution":{"observed_at":"2026-08-06T19:31:21.883499Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2209.11895","last_updated":"2022-09-24T00:43:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-09-24T00:43:19Z","title":"In-context Learning and Induction Heads","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.11895","snapshot_observed_at":"2026-08-06T19:31:22.053685Z","title":"Olsson, N","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:22.053685Z"},"links":{"cited_paper":"/paper/2209.11895","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:390154b2fb5c6576e8c57147f70fd9978bbcc39321a09897625843cbfe38c492","observation_id":"ecb45ebc-fede-4a06-9a48-ddf234580563","resolution":{"observed_at":"2026-08-06T19:31:22.053685Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.704513Z","title":"Radhakrishnan, D","venue":null,"work_id":"acc8aec0-632a-4012-9687-275c784cab3e","year":2024},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:22.138439Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:034f98fc0dd10ad84a360c3ffccc9ced0106e16f1232e11566590d7ca908454f","observation_id":"0ca2b527-e84b-4955-8065-e6132fc15d43","resolution":{"observed_at":"2026-08-06T19:31:24.708689Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.690510Z","title":"Radhakrishnan, M","venue":null,"work_id":"75ebcf0d-e818-4a2c-b187-70aaa49ea2be","year":2025},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:22.239240Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:5081630229ce23989fc05865039891b1446926fd1ddc78855cce9bf0b7b920cc","observation_id":"24902931-396d-4ebb-ae42-dfefc601ed95","resolution":{"observed_at":"2026-08-06T19:31:24.694679Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.674898Z","title":null,"venue":null,"work_id":"601b52b6-c7e6-4569-8467-7a96af810a55","year":1969},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:22.355074Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:db7865216da116fe1d9c74da35aca60d48ec94dee4f31572164b95d2edcd0189","observation_id":"9a72e09f-4fe4-4ee2-98d2-82d05b4f194d","resolution":{"observed_at":"2026-08-06T19:31:24.679642Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.660584Z","title":null,"venue":null,"work_id":"fbf25fa4-81d2-4c69-a687-649f0a460df1","year":2014},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:22.438935Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:b4a4d6688bd856436230855bff4b06f031855f746a5230a272dff0f0aa1f3cb8","observation_id":"7bb330c8-4970-42e6-847e-da508dcfbfc8","resolution":{"observed_at":"2026-08-06T19:31:24.665209Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.645502Z","title":"Schölkopf","venue":null,"work_id":"f86957e3-aee1-49ce-8ddd-ea0bb6cd9226","year":2002},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:22.618654Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:ebc5a12a8e7578de57bead62fe5e0bb41fc1ca1a1825c8076aa723d930133efc","observation_id":"fbf637da-08f3-4cf0-b627-d1e2759742b0","resolution":{"observed_at":"2026-08-06T19:31:24.650123Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.627904Z","title":"Soudry, E","venue":null,"work_id":"9455cee2-8567-4c16-a927-deedc6c31b1a","year":2018},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:22.728679Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:da2e57f4c6e07c0e9ebccf11c63d2b9f494fa872cbfe1d45d0dfae8a65216261","observation_id":"fcaf784b-b920-43de-94da-14a76e9b50b3","resolution":{"observed_at":"2026-08-06T19:31:24.634050Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.612620Z","title":"Stewart, F","venue":null,"work_id":"531816a2-6909-4218-bb2a-169eeca0d71c","year":2023},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:22.817942Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:c1fe88955f5903b8243fadfc9e6f79521a18cb1b0e0929becc6749b497295152","observation_id":"578f007b-5beb-46ed-ae21-ba15c00ceea1","resolution":{"observed_at":"2026-08-06T19:31:24.616876Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.596085Z","title":"Thompson","venue":null,"work_id":"5de6b7e4-5d84-4dc9-afda-8fedbc1ac209","year":1976},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:22.979868Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:2681be709da50bc2d5dcc48928ea062d11044ce0c135652814230c02f5c772f1","observation_id":"2fee3bd5-31e5-4c92-a55e-04e0eb049272","resolution":{"observed_at":"2026-08-06T19:31:24.600429Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.582052Z","title":"Woodworth, S","venue":null,"work_id":"bf3aaa6d-7156-496e-bfcc-30ce0a885d09","year":2020},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:23.157017Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:466080895ef8fb707e5e64298c6aa63700a887f355c7c803f2ecb6c1024f988e","observation_id":"620f0ca0-1e19-4cef-863c-21cb30acd2f2","resolution":{"observed_at":"2026-08-06T19:31:24.586352Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:23.296676Z","title":"Zangrando, P","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:23.296676Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:bc583f9d8d2600c6ad15d6ad41d13160fbe1fdf9d69716a9e988d9d5e36ae720","observation_id":"b304a0c9-3587-4c97-bdcc-9f5161999290","resolution":{"observed_at":"2026-08-06T19:31:23.296676Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:23.400916Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:23.400916Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:4ad7776b6aa1d2ca57d83e16675c6b1bb5d02f70f03e6c48ffdc245ab01cc069","observation_id":"cd342bf4-0045-4985-acf7-0e0c7e1e5b85","resolution":{"observed_at":"2026-08-06T19:31:23.400916Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.04815","last_updated":"2024-06-06T03:57:32Z","snapshot_observed_at":"2026-08-13T11:22:14.671128Z","submitted_at":"2023-06-07T22:37:11Z","title":"Catapults in SGD: spikes in the training loss and their impact on generalization through feature learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.04815","snapshot_observed_at":"2026-08-06T19:31:23.560693Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:23.560693Z"},"links":{"cited_paper":"/paper/2306.04815","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:6a04efae12207fef08aa44610ac1f7e4530edb73777bb4cb02811c2bbb7fa28a","observation_id":"af7e1ca8-b352-49da-a3f4-621f29fa83df","resolution":{"observed_at":"2026-08-06T19:31:23.560693Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.568140Z","title":"Ziyin, I","venue":null,"work_id":"8a915652-1ba4-4855-970d-d2107ce7c743","year":2025},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:23.719685Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:4649221c4d47b4183be134815c1cc2321156588264b3fe1eed2d9e8663077743","observation_id":"706c1eac-8e9a-47dd-beec-0b0a644a6a20","resolution":{"observed_at":"2026-08-06T19:31:24.572126Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.552190Z","title":"Ziyin, B","venue":null,"work_id":"4042745e-86e7-4c4c-a0c9-7414e1a7baca","year":2022},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:23.888184Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:07f4c51f9a0c91e8f2e9485a89d64722b5d78c2dfe88b8adb526be92f9eb7f43","observation_id":"11db1100-1a4e-49fe-a9b0-9225509e8ed9","resolution":{"observed_at":"2026-08-06T19:31:24.557660Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-13T16:40:24.297937Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations"},"reference_resolution":{"displayed":53,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":29,"verified_exact":0,"verified_fuzzy":24},"total_outbound_references":53},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 53 of 53 outbound references and 1 inbound Pith citation observation for arXiv:2507.05644."}