{"as_of":"2026-08-09T16:38:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:775b13c40b1a969a3bb54261cf195f4ee1b6295fd063c1f080dc1e522947366d","coverage":[{"denominator":63,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":63,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-28T07:02:26.496063Z","state":"measured"},{"denominator":63,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":63,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2606.04476/citation-record","integrity":"/paper/2606.04476/integrity","json":"/paper/2606.04476/citation-record.json","paper":"/paper/2606.04476"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Fine-grained analysis of optimization and generalization for overparameterized two-layer neural networks","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:7e30d404e138da8a7c20c895fe840c054f4d0ee4e70bcbaa9409d8024663ae4f","observation_id":"57185953-a4df-4a34-9bd0-a1b5c1925799","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"High-dimensional asymptotics of feature learning: How one gradient step improves the representation.Advances in Neural Information Processing Systems, 35:37932–37946, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:47c9af4dacf58f816a0ac15c3ed519849ca29d06282fb4cb064b607fa26c02ad","observation_id":"9e95d3e5-d7e5-4128-93dc-b149ac67772a","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Neural networks and principal component analysis: Learning from examples without local minima.Neural Networks, 2(1):53–58, 1989","venue":null,"work_id":null,"year":1989},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:d5539a9fcb46ad3f1a39e2964f6916565e304845e5f61d8ca85b86afb8a76709","observation_id":"5af30a62-6788-470a-9ebb-ac848617fadf","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Simplicity bias and optimization threshold in two-layer relu networks, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:490a2bc550bd6d514b1f69c4a4307bd9b88521a10730b3ff35a3053b7cec81b2","observation_id":"a06d5dc0-6eb1-49a4-95b3-36611bbc7553","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"A Bennett concentration inequality and its application to suprema of empirical processes.C","venue":null,"work_id":null,"year":2002},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:7dd6196399ed88308a821f2975d47ba34697df8f78d47075c5fcfe101d3bed9d","observation_id":"fd2508e3-f5b7-4d29-b010-934b8c2201c5","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Globally optimal gradient descent for a convnet with gaussian inputs, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:cb40cfa019f8e2062d0a363e831b8e2e8d809bdbfd1deae55b150c1daadfad52","observation_id":"9f293b8b-51e8-40e2-a300-fb66cab3cd4d","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Candès, Xiaodong Li, and Mahdi Soltanolkotabi","venue":null,"work_id":null,"year":1985},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:4d1730b4330a9f09e30c37f51210a09b9385916323aa3424b0fe7489229c99c6","observation_id":"4b939cc3-a74d-4ed7-9739-dd88efbc24e1","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Nonconvex rectangular matrix completion via gradient descent without ℓ2,∞regularization.IEEE Trans","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:fd4004178db0ed36848459a7a2097e4a3c8ac19e517511033e6e9732afb14de3","observation_id":"980bddcc-1df7-428d-a3d5-501c69a7db43","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:3de2f43050cc5083862f2ebd2a1996f254ccea6e65f982a1f6a2d5abbfa70f05","observation_id":"246ee337-c9c0-4f73-a3cb-d275e5d9e9ac","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Learning a neuron by a shallow relu network: Dynamics and implicit bias for correlated inputs, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:ac2c3a895d93724ab80f2646b89b00e5cac4787b55383bec86c5fc4be4916b07","observation_id":"495f760b-4bd6-4dc8-80b6-60276cb720f2","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"On lazy training in differentiable programming.Advances in Neural Information Processing Systems, 32:2937–2947, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:e22b8659ba88af0687ab09d651a6072c09fd91448ceded443983bd5c556f7336","observation_id":"3dcff423-d702-40bf-a2b4-6dc350c73534","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Neural networks can learn representations with gradient descent","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:2f2001af64f0611bbca70f7d8979e18961b8dc0b0a1d031904b2ae711a2cb856","observation_id":"8a8ea038-8389-45e7-ad9b-95541430fdfa","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Toward deeper understanding of neural networks: The power of initialization and a dual view on expressivity.Advances in neural information processing systems, 29, 2016","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:b669febc728f66c7e40b0069121e65c314dfdaf7a174961048e91eb3a8046db1","observation_id":"6455b47c-e0a3-41c0-afac-781577401f66","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Gradient descent finds global minima of deep neural networks","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:f8591a80b26dc32665aaa336e3029ce038577db17c3d57e8755451eefe225e25","observation_id":"3a49fd5b-e0c6-4a37-a088-4249ba4e696a","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1810.02054","last_updated":"2019-02-05T01:59:59Z","snapshot_observed_at":"2026-08-08T10:07:26.425335Z","submitted_at":"2018-10-04T04:47:47Z","title":"Gradient Descent Provably Optimizes Over-parameterized Neural Networks","version":2},"cited_work":{"arxiv_id":"1810.02054","doi":null,"metadata_source":"pith","pith_arxiv_id":"1810.02054","snapshot_observed_at":"2026-07-10T08:56:58.947053Z","title":"Gradient Descent Provably Optimizes Over-parameterized Neural Networks","venue":"cs.LG","work_id":"f8869a2d-5e6f-4ab3-8826-04c3b0855136","year":2018},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"cited_paper":"/paper/1810.02054","citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:03dd68a7755d794580e01e09c64feb2f70b4b41c0ca6fac7f83cd55fce4d4c9d","observation_id":"6100d680-3410-4d10-9465-618c5a1d845d","resolution":{"observed_at":"2026-07-02T07:16:44.874627Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Humus-net: Hybrid unrolled multi-scale network architecture for accelerated mri reconstruction.Advances in Neural Information Processing Systems, 35:25306–25319, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:1d37c0bf862c575940662566ae733b25ffb677870526233c7a1ec167c3d827d5","observation_id":"71eaa516-27f7-4e67-9354-0dc33d66e49b","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":null,"venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:fb4c3dfe0fb7fe9f7379e303f0e9c630563746e20879af60be4cd847a4f7de61","observation_id":"dba5a5cb-020e-4e10-a5af-4802b8957a44","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Matrix completion has no spurious local minimum.Advances in Neural Information Processing Systems, 29:2973–2981, 2016","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:9b18d253a3219bc5cce07872ef9f07f25489d058fabf51c67f708977341205a6","observation_id":"3181f5c1-d51c-4561-a42a-a51b3347f1a0","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.13409","last_updated":"2021-11-09T21:01:40Z","snapshot_observed_at":"2026-07-06T09:32:08.181622Z","submitted_at":"2020-06-24T01:03:31Z","title":"When Do Neural Networks Outperform Kernel Methods?","version":2},"cited_work":{"arxiv_id":"2006.13409","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2006.13409","snapshot_observed_at":"2026-07-02T07:16:44.876051Z","title":"When do neural networks outperform kernel methods?","venue":null,"work_id":"34422d98-20cb-484b-9849-e75d4a8e91d7","year":2006},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"cited_paper":"/paper/2006.13409","citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:bb5aab58ad32ef4fcb2aa9d40a6877ffa4ec65341af107c6dfd7c8f2eb077ba5","observation_id":"72ce50a2-e2d9-40fa-89d7-0a8d40216066","resolution":{"observed_at":"2026-07-02T07:16:44.878394Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Phase retrieval under a generative prior.Advances in Neural Information Processing Systems, 31, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:2e0efb8113bc92d1b73add2d3ecd5542fc42eb2c0be43939d37362a496f4b4f9","observation_id":"0dc9521a-b450-4cc3-85f0-f74b54d3421f","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Neural tangent kernel: Convergence and generalization in neural networks","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:dd10f616d6c0b74a1f9b640a521e37820bfdf281e072cacfff50c6dc6b5b1532","observation_id":"e4d2285b-ddad-4789-87e9-e14b879a8ab2","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Gradient descent aligns the layers of deep linear networks, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:647390713672d1359284cc5037e24fdb1bd30815f7551c47ed51a06846829110","observation_id":"8f27dee3-d753-4c12-9190-cec6ed2a350d","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Kakade, and Michael I","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:634db5691d26058765372cec8e4a19a63c405d1b12fcedbf6fd15c0f9213d767","observation_id":"7684b687-cc64-4410-a15b-92adeb0fb4ed","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Deep convolutional neural network for inverse problems in imaging.IEEE transactions on image processing, 26(9):4509–4522, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:f8b55efd0bb37c9b8b28def54fe19bad4e16675761a66743a4449c16891f5199","observation_id":"50d7a600-864b-4dc7-ae3c-f6a363bbc4bc","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Photo-realistic single image super-resolution using a generative adversarial network","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:58099a0dfe1dac010a266c5897c402b777b21d0decfd2a8d96bc6ffa93dd96d5","observation_id":"84407dcb-c0f7-44fd-94e6-587873124502","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01581","last_updated":"2024-12-22T09:59:09Z","snapshot_observed_at":"2026-08-04T02:53:19.400873Z","submitted_at":"2024-06-03T17:56:58Z","title":"Neural network learns low-dimensional polynomials with SGD near the information-theoretic limit","version":2},"cited_work":{"arxiv_id":"2406.01581","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.01581","snapshot_observed_at":"2026-07-02T07:16:44.862552Z","title":"arXiv preprint arXiv:2406.01581 , year=","venue":null,"work_id":"5e2e7201-43e5-4599-8ed3-6237db94f737","year":2024},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"cited_paper":"/paper/2406.01581","citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:0fa8d382c81870274f34ff65b0ada60a1572d26e7f1ddb5e7cf165c20f1f0e4f","observation_id":"aba5695e-beb9-4e0d-85fa-025b9a9eb587","resolution":{"observed_at":"2026-07-02T07:16:44.864272Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.10322","last_updated":"2025-03-01T04:06:51Z","snapshot_observed_at":"2026-07-06T19:32:56.002390Z","submitted_at":"2024-10-14T09:28:32Z","title":"Feature Averaging: An Implicit Bias of Gradient Descent Leading to Non-Robustness in Neural Networks","version":2},"cited_work":{"arxiv_id":"2410.10322","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.10322","snapshot_observed_at":"2026-07-02T07:16:44.865516Z","title":"Feature averaging: An implicit bias of gradient descent leading to non-robustness in neural networks.arXiv preprint arXiv:2410.10322, 2024","venue":null,"work_id":"42ffa6ad-d83c-4291-a2b3-8525cfc11752","year":2024},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"cited_paper":"/paper/2410.10322","citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:28e8b2f1370a0f125c9855c13f451232a15f70dd428fdd3cf14bbb53e97af4ac","observation_id":"d685ddc0-19cf-46b0-b72f-413e7189bb81","resolution":{"observed_at":"2026-07-02T07:16:44.867756Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Rapid, robust, and reliable blind deconvolution via nonconvex optimization.Appl","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:91c25f415a058a615187f50bcbb8b503a8ce1fc012f639a0cf7f37710bbe7548","observation_id":"41d97781-70de-458b-83df-456bfe6e1067","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Regularized gradient descent: a non-convex recipe for fast joint blind deconvolution and demixing.Inf","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:3816aef43317fe6b1630678e01e5a43a78cf7d1dceab199311674a6cfe7cb667","observation_id":"ea752827-4461-4d1d-bd18-7ed2b773024a","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Implicit regularization in nonconvex statistical estimation: gradient descent converges linearly for phase retrieval, matrix completion, and blind deconvolution.Found","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:e58743893d7f0670a4e7f370087307d68f2abf81a5317e7da869230b22e376fd","observation_id":"47afa560-2e90-4270-8211-3a8e90dc06ee","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":null,"venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:41312911437a80e074b951d712fab0e288dcb6021bb713f13ef8237f7fb23aaa","observation_id":"3f459658-b3f8-4aa4-88a9-7e626bcf57a2","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":null,"venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:8817059d90ef480a1a21775c970395c34a53c63fad92f67d6b0b1d94c8b3e462","observation_id":"fd2fd0af-e3fe-4d7c-bdb5-3776b8bf816e","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1910.01635","last_updated":"2019-10-03T17:56:10Z","snapshot_observed_at":"2026-07-06T08:26:44.209184Z","submitted_at":"2019-10-03T17:56:10Z","title":"A Function Space View of Bounded Norm Infinite Width ReLU Nets: The Multivariate Case","version":1},"cited_work":{"arxiv_id":"1910.01635","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1910.01635","snapshot_observed_at":"2026-07-02T07:16:44.880017Z","title":"A function space view of bounded norm infinite width ReLu nets: The multivariate case","venue":null,"work_id":"a0755800-5265-46f8-aa11-202d179af288","year":1910},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"cited_paper":"/paper/1910.01635","citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:434bdb4d7c0b33396e7009d31634cb2b5b25857d2e9e256f0796f666f729c386","observation_id":"b635dc95-e41e-498d-be21-720f3be637fd","resolution":{"observed_at":"2026-07-02T07:16:44.881944Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Overparameterized nonlinear learning: Gradient descent takes the shortest path? InInternational Conference on Machine Learning, pages 4951–4960","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:2d995755ce1f839b65a6b2e617e522523d287d72b005c48e24f3997233cc1331","observation_id":"f904e8fd-631a-4cb3-b7b5-acd293d1f1e8","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Towards moderate overparameterization: global convergence guarantees for training shallow neural networks.IEEE Journal on Selected Areas in Information Theory, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:6e8164b4e7d06ba9436b96a9d37bdfaff91091faff6c02bab1e794ad11553bd0","observation_id":"1c3ec22f-c3a4-4e54-9ef4-1b4026774f2f","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Grokking: Generalization beyond overfitting on small algorithmic datasets, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:4d8fb429d6464780aa152a042d40dbc51fbe4625c94ea61fe698f19366c3d16a","observation_id":"0d7aad34-f164-4b07-9ed0-af519f7c0fe7","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Non-convex learning via stochastic gradient langevin dynamics: a nonasymptotic analysis","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:30213cb0eec2e7c014b7d827bceb4b721156ae90b8a3948d9511ac1d51f41cb0","observation_id":"63e86a6b-9d3d-4c52-9451-a3e42b22ff0d","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:8c8f4ba8344580ebc5a7cd8448293e9f0d81dfa4e11c6b121d7080365c377bff","observation_id":"fbd3d35f-7476-41dc-956e-1e0e22cb58d1","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Learning relus via gradient descent.Advances in neural information processing systems, 30, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:f5d21a32f06fd92ca6856e67081f562880e79a9a5d9817afd835b0335ba994e3","observation_id":"87222743-2ab5-4b19-910e-82b68662a8ec","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Theoretical insights into the optimization landscape of over-parameterized shallow neural networks.IEEE Transactions on Information Theory, 65(2):742–769, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:264519f37aa4bb34694dff7dfad4eb9f1f81360e66d27823779aca71e93fb5d9","observation_id":"7227e63c-7854-4dbc-97cd-1181472fb6a5","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Implicit balancing and regularization: Generalization and convergence guarantees for overparameterized asymmetric matrix sensing","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:48b7e068061ee0292c4c8db30dcc0baa6993845a5a9eb52a4285b19cea3eae2e","observation_id":"756b71e2-3a6b-405b-854c-1dda97b2b8a3","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"End-to-end variational networks for accelerated mri reconstruction","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:598d2b1fde06a611d9149938528837c2fce5766644cb359cc84fa842419c7990","observation_id":"78b89da2-4051-4c69-aeb5-2ae94f44dbe0","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Small random initialization is akin to spectral learning: Optimization and generalization guarantees for overparameterized low-rank matrix reconstruction","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:1593606c0a78fe1e84616004589face80b6a04edbb2ee96431271e62c0b07a99","observation_id":"ebc90a49-d7cf-457a-9d41-fd359d60b2c1","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1510.06096","last_updated":"2016-04-23T00:18:06Z","snapshot_observed_at":"2026-07-06T04:33:47.074419Z","submitted_at":"2015-10-21T00:59:23Z","title":"When Are Nonconvex Problems Not Scary?","version":2},"cited_work":{"arxiv_id":"1510.06096","doi":null,"metadata_source":"pith","pith_arxiv_id":"1510.06096","snapshot_observed_at":"2026-07-02T11:46:55.219319Z","title":"When Are Nonconvex Problems Not Scary?","venue":"math.OC","work_id":"cc848323-59bd-4de0-ac58-8d7113f9facf","year":2015},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"cited_paper":"/paper/1510.06096","citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:feb1d57d9fa5d633dc6f3cc0d435cff012e90a11c9a6a984f088341a6a09b3bd","observation_id":"62a2fc1b-a708-4ff5-9826-387102365028","resolution":{"observed_at":"2026-07-02T07:16:44.871423Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"A geometric analysis of phase retrieval.Found","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:30b8ca29ac245e02dfc395209697f7daf27664b13322c01bd4dc16617a119a29","observation_id":"61f1fb50-214e-4390-8ba2-fb0daddf41ab","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Low-rank solutions of linear matrix equations via procrustes flow","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:d2e742a48bd0e2c4adb77c5bade5ca78732cfbc12037ebf3ba3b8257347db382","observation_id":"d35d7fff-b8d6-4bb9-b567-9c4c4fe74d5f","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Van der Vaart and Jon A","venue":null,"work_id":null,"year":1996},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:cb8e7c2ad4c3f4f5d4b98b7a9a3f4e6f5fc4aa1ec664fe79f5752b616df6aeb8","observation_id":"ddc13a1c-d38b-4414-998e-18e57f19c474","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Learning a single neuron with bias using gradient descent, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:9e9279b84659dba730661f0b473257413a5f55b32c2f62a06bd487f2023368f1","observation_id":"e375c9c7-d574-4f1f-aad7-4ed5bbecab58","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Wainwright.High-dimensional statistics","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:5b37ba4e596f13edde2cf719dd6ba63fe6f9b5edb3a89b2ab9fd6d4678ef348d","observation_id":"dcdc7dde-9c4c-467c-864e-1792469da515","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Image inpainting via generative multi-column convolutional neural networks.Advances in neural information processing systems, 31, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:e1777309908de8079aa37dd535bb3a9cc31c3d47d31eaeddd8382bbb1dbbc769","observation_id":"da3bdf4a-ac7c-46dc-b313-2006c733f1dd","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Large learning rate tames homogeneity: Convergence and balancing effect, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:a719464fcca3825ae69ee97f54bc2d1816a95204b70adb9c605fb1f34a61f8a7","observation_id":"8c82d051-1ac8-4237-b812-fa624d2c437f","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Good regularity creates large learning rate implicit biases: edge of stability, balancing, and catapult, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:ff7df529a01c934d0bc7fd42bd95af2174fcbd57059d984b74b40734ff77f4bf","observation_id":"81fac2b4-4cf9-49ca-8370-8169aea3f154","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:f9cece3439d4aacf1863de690915a970ef0b150cda274e4570e4f4494ea71422","observation_id":"e96f5f4a-67f1-4a56-a317-602b0775b5f5","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:04e378c03d39d8fd64e3f6d37c4333dc41ce1208335c0e3f70750ed8e09558e2","observation_id":"596557ce-5e47-4b9c-9d62-32156b5a4958","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Learning a single neuron with gradient methods, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:7e9f571ed7b38302c08c1efbee80560edcf8091f858165367217c552b1de40b6","observation_id":"19f688ef-6516-4f17-975e-198422718046","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Zhang, Somayeh Sojoudi, and Javad Lavaei","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:3c1b44b9d58d8d1d932f3ae2a50c2dc6a5ce8e05847dedd0b9f6be7eb80d558b","observation_id":"fda81545-33ef-4469-9b13-203c01e3a889","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Learning one-hidden-layer relu networks via gradient descent, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:8ff5a1b6bca9207e711ba52b35ab028be36f0aff5e666e14839481a927195781","observation_id":"2a55c107-15df-49cf-b214-9510fd40dc36","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"A hitting time analysis of stochastic gradient langevin dynamics","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:3e4bcd7c44125534a048b5f1111efe5ff8d2f7daa20f3bf5dbadb12b2736d0bb","observation_id":"359292a2-35e6-4976-b1ae-547ebe25bda5","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Bartlett, and Inderjit S","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:cd5b435a36fcd08c0751c821d0b11a8def1b890ff6f4714cd9ef23cda20d5cd7","observation_id":"2efa4b6f-85cc-4342-adc8-27100279b3e8","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"How gradient descent balances features: A dynamical analysis for two-layer neural networks","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:28d7e4eaf981f09228b48ed825719aa2773325f9025f06f4558258372487eb8b","observation_id":"a4868aed-ff9f-4e2b-aa3d-003293fc4c6d","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"Here we use the fact that µv2 1 ≤c 0c2 2 ≤ 1","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:d5d228f7bf78d8318db6a8bfa027f14a2a2ff484a05ec0da8d89505a09211ef6","observation_id":"f8acf33b-3d50-4302-b93e-ea5f66a40076","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"As a result, the sum v1 +∥w 1∥ will increase by a factor 1 + 1 8 µ∥a∥ as long as v1 ∥w1∥< 1 4 ∥a∥","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:6d53458e893124dd28702126108a85bb8a17b031c0121ac0bfc1b083eb2a1a31","observation_id":"b9b5573f-e965-42c3-b61f-02dd1da4eb8a","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T07:02:26.496063Z","title":"v(τ) 1 v(τ) 2 #! W (τ) −W 2 F . 45 By further upper bounding the right-hand side using the fact that the ReLU activation is 1-Lipschitz, we have L θ(T) ≤ diag","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:26.496063Z"},"links":{"citing_paper":"/paper/2606.04476"},"observation_digest":"sha256:bb13505dd4442250f2344abfeae465dd05986edffdb62211955ee00b14daefac","observation_id":"8a070a42-252e-4975-b378-f648de45ba4a","resolution":{"observed_at":"2026-06-28T07:02:26.496063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2606.04476","last_updated":"2026-06-03T05:44:30Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-08T11:18:56.709201Z","submitted_at":"2026-06-03T05:44:30Z","title":"When Both Layers Learn: Training Dynamics of Representing Linear Models via ReLU Networks"},"reference_resolution":{"displayed":63,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":57,"verified_exact":6,"verified_fuzzy":0},"total_outbound_references":63},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 63 of 63 outbound references and 0 inbound Pith citation observations for arXiv:2606.04476."}