{"as_of":"2026-08-11T18:06:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:421cef9ebb99656a8159760e17073ec39f0c5c62f38529efb630169af92922d5","coverage":[{"denominator":26,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":26,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T20:26:43.530125Z","state":"measured"},{"denominator":27,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":27,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-29T13:20:54.303605Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-06-29T13:23:28.017867Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.09137","last_updated":"2025-01-21T03:05:17Z","snapshot_observed_at":"2026-08-11T05:49:13.700840Z","submitted_at":"2025-01-15T20:43:36Z","title":"Gradient Descent Converges Linearly to Flatter Minima than Gradient Flow in Shallow Linear Networks","version":2},"cited_work":{"arxiv_id":"2501.09137","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.09137","snapshot_observed_at":"2026-06-29T13:23:28.017867Z","title":"Gradient descent converges linearly to flatter minima than gradient flow in shallow linear networks.arXiv preprint arXiv:2501.09137, 2025","venue":null,"work_id":"064b4b99-f5ac-4a76-947b-3ec69d8aa572","year":2025},"citing_paper":{"arxiv_id":"2605.29152","last_updated":"2026-05-27T22:30:39Z","snapshot_observed_at":"2026-08-06T17:53:39.851226Z","submitted_at":"2026-05-27T22:30:39Z","title":"Do Deep Networks Forget Initialization? A Forgetting-Time View of Practical Inductive Bias","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-06-29T13:20:54.303605Z"},"links":{"cited_paper":"/paper/2501.09137","citing_paper":"/paper/2605.29152"},"observation_digest":"sha256:89e3d754acd6f57f75f3253ded0c128a7b78dd410be13617339b73bec167fb7d","observation_id":"f1271a0b-d08e-4163-8536-08678091a4bd","resolution":{"observed_at":"2026-06-29T13:23:28.020113Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2501.09137/citation-record","integrity":"/paper/2501.09137/integrity","json":"/paper/2501.09137/citation-record.json","paper":"/paper/2501.09137"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:26:43.412162Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.09137","last_updated":"2025-01-21T03:05:17Z","snapshot_observed_at":"2026-08-11T05:49:13.700840Z","submitted_at":"2025-01-15T20:43:36Z","title":"Gradient Descent Converges Linearly to Flatter Minima than Gradient Flow in Shallow Linear Networks","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-10T20:26:43.412162Z"},"links":{"citing_paper":"/paper/2501.09137"},"observation_digest":"sha256:29eac25fe14354e9416a4e18225bb38c93b1ad54ce8bf18a517726606bcfea3d","observation_id":"a3c0857e-9576-4883-b16a-138151270d33","resolution":{"observed_at":"2026-08-10T20:26:43.412162Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:26:43.937829Z","title":"T., Suarez, F., and Zhang, Y","venue":null,"work_id":"429bd099-d1bb-4db7-9afd-54a1b24b90c2","year":2024},"citing_paper":{"arxiv_id":"2501.09137","last_updated":"2025-01-21T03:05:17Z","snapshot_observed_at":"2026-08-11T05:49:13.700840Z","submitted_at":"2025-01-15T20:43:36Z","title":"Gradient Descent Converges Linearly to Flatter Minima than Gradient Flow in Shallow Linear Networks","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-10T20:26:43.417092Z"},"links":{"citing_paper":"/paper/2501.09137"},"observation_digest":"sha256:768851816f01e1fd36704b528fb0ea1426692107b61e551ffc447e847751a8db","observation_id":"c5f48a3d-e7f8-4ddc-a62f-e1616fe57885","resolution":{"observed_at":"2026-08-10T20:26:43.941693Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1810.02281","last_updated":"2019-10-26T06:58:22Z","snapshot_observed_at":"2026-08-06T22:39:11.101764Z","submitted_at":"2018-10-04T15:53:32Z","title":"A Convergence Analysis of Gradient Descent for Deep Linear Neural Networks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1810.02281","snapshot_observed_at":"2026-08-10T20:26:43.421051Z","title":"A Convergence Analysis of Gradient Descent for Deep Linear Neural Networks , October 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2501.09137","last_updated":"2025-01-21T03:05:17Z","snapshot_observed_at":"2026-08-11T05:49:13.700840Z","submitted_at":"2025-01-15T20:43:36Z","title":"Gradient Descent Converges Linearly to Flatter Minima than Gradient Flow in Shallow Linear Networks","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-10T20:26:43.421051Z"},"links":{"cited_paper":"/paper/1810.02281","citing_paper":"/paper/2501.09137"},"observation_digest":"sha256:a75d13c6d7ef2832739c38b7adb3ad315dc7de38a71551f09aeba059318aa2af","observation_id":"c5b2b8e0-c4fb-4cb4-b295-8567443c7197","resolution":{"observed_at":"2026-08-10T20:26:43.421051Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:26:43.921125Z","title":"P., Selman, B., and Weinberger, K","venue":null,"work_id":"c545dd5c-561b-425e-804a-d78c6cf25014","year":2018},"citing_paper":{"arxiv_id":"2501.09137","last_updated":"2025-01-21T03:05:17Z","snapshot_observed_at":"2026-08-11T05:49:13.700840Z","submitted_at":"2025-01-15T20:43:36Z","title":"Gradient Descent Converges Linearly to Flatter Minima than Gradient Flow in Shallow Linear Networks","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-10T20:26:43.426425Z"},"links":{"citing_paper":"/paper/2501.09137"},"observation_digest":"sha256:8c9504f790f1617c2509cc0c51712c1c3def363e679dc1a90a4aba8ebf784ab0","observation_id":"da037326-655f-489c-9d80-f24bc6feaf81","resolution":{"observed_at":"2026-08-10T20:26:43.928142Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1606.04838","last_updated":"2018-02-08T20:40:22Z","snapshot_observed_at":"2026-08-09T09:43:11.255145Z","submitted_at":"2016-06-15T16:15:53Z","title":"Optimization Methods for Large-Scale Machine Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1606.04838","snapshot_observed_at":"2026-08-10T20:26:43.430882Z","title":"E., and Nocedal, J","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2501.09137","last_updated":"2025-01-21T03:05:17Z","snapshot_observed_at":"2026-08-11T05:49:13.700840Z","submitted_at":"2025-01-15T20:43:36Z","title":"Gradient Descent Converges Linearly to Flatter Minima than Gradient Flow in Shallow Linear Networks","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-10T20:26:43.430882Z"},"links":{"cited_paper":"/paper/1606.04838","citing_paper":"/paper/2501.09137"},"observation_digest":"sha256:7fc83a0e01391d943d28d19e22f5c25ed7e972abb3a22ccfea192005ba4b7c97","observation_id":"e59cfc15-b3c9-487d-88e6-bf090f02520f","resolution":{"observed_at":"2026-08-10T20:26:43.430882Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:26:43.902738Z","title":"and Bruna, J","venue":null,"work_id":"514aea08-0892-408a-9cd8-d956b3950c1d","year":2023},"citing_paper":{"arxiv_id":"2501.09137","last_updated":"2025-01-21T03:05:17Z","snapshot_observed_at":"2026-08-11T05:49:13.700840Z","submitted_at":"2025-01-15T20:43:36Z","title":"Gradient Descent Converges Linearly to Flatter Minima than Gradient Flow in Shallow Linear Networks","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-10T20:26:43.435109Z"},"links":{"citing_paper":"/paper/2501.09137"},"observation_digest":"sha256:04faecd60c668fa8b9d87446eed17cdbe3762c5593cd2d813dc299cdfe308445","observation_id":"f5d9d63a-5642-4d59-ba11-4f91cf4fb02d","resolution":{"observed_at":"2026-08-10T20:26:43.908686Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:26:43.883348Z","title":"Z., and Talwalkar, A","venue":null,"work_id":"77954181-3d87-4eba-811d-e70de5671c63","year":2021},"citing_paper":{"arxiv_id":"2501.09137","last_updated":"2025-01-21T03:05:17Z","snapshot_observed_at":"2026-08-11T05:49:13.700840Z","submitted_at":"2025-01-15T20:43:36Z","title":"Gradient Descent Converges Linearly to Flatter Minima than Gradient Flow in Shallow Linear Networks","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-10T20:26:43.440119Z"},"links":{"citing_paper":"/paper/2501.09137"},"observation_digest":"sha256:738929d696bb15820b8cafaa0770b3c8f0bb0bfefdc6fed623fe498033061b99","observation_id":"5ec7451e-75d1-4dd3-9dad-51763c0f20e3","resolution":{"observed_at":"2026-08-10T20:26:43.889619Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:26:43.868959Z","title":"Implicit Regularization of Discrete Gradient Dynamics in Linear Neural Networks","venue":null,"work_id":"16984dbf-fc15-4034-9971-d3946a47796f","year":2019},"citing_paper":{"arxiv_id":"2501.09137","last_updated":"2025-01-21T03:05:17Z","snapshot_observed_at":"2026-08-11T05:49:13.700840Z","submitted_at":"2025-01-15T20:43:36Z","title":"Gradient Descent Converges Linearly to Flatter Minima than Gradient Flow in Shallow Linear Networks","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-10T20:26:43.445461Z"},"links":{"citing_paper":"/paper/2501.09137"},"observation_digest":"sha256:5ae04277081c7776842635d4c0dc72c442e30768e52a0dbf25082e1884131270","observation_id":"98fce254-a364-4024-8a8d-cf374f5f1177","resolution":{"observed_at":"2026-08-10T20:26:43.873290Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:26:43.850140Z","title":"and Schmidhuber, J","venue":null,"work_id":"c410373d-b8eb-40d0-93ee-fd675d3a8a00","year":1997},"citing_paper":{"arxiv_id":"2501.09137","last_updated":"2025-01-21T03:05:17Z","snapshot_observed_at":"2026-08-11T05:49:13.700840Z","submitted_at":"2025-01-15T20:43:36Z","title":"Gradient Descent Converges Linearly to Flatter Minima than Gradient Flow in Shallow Linear Networks","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-10T20:26:43.449369Z"},"links":{"citing_paper":"/paper/2501.09137"},"observation_digest":"sha256:976ca09d9a3f67482413e4737243c9e3eed9700442b1be50f9b7efe2daf93e91","observation_id":"a66da8c3-f9f6-4ce4-8668-f04cfd0b3d25","resolution":{"observed_at":"2026-08-10T20:26:43.857854Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2002.09572","last_updated":"2020-02-21T22:55:51Z","snapshot_observed_at":"2026-08-09T11:01:21.008891Z","submitted_at":"2020-02-21T22:55:51Z","title":"The Break-Even Point on Optimization Trajectories of Deep Neural Networks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2002.09572","snapshot_observed_at":"2026-08-10T20:26:43.454094Z","title":"The Break - Even Point on Optimization Trajectories of Deep Neural Networks , February 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2501.09137","last_updated":"2025-01-21T03:05:17Z","snapshot_observed_at":"2026-08-11T05:49:13.700840Z","submitted_at":"2025-01-15T20:43:36Z","title":"Gradient Descent Converges Linearly to Flatter Minima than Gradient Flow in Shallow Linear Networks","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-10T20:26:43.454094Z"},"links":{"cited_paper":"/paper/2002.09572","citing_paper":"/paper/2501.09137"},"observation_digest":"sha256:eb3d543fb89342676bfaa9a9ecd311de8308a3d47a0cc57c7b28e7d07c1de11d","observation_id":"836d0d80-4c5a-441e-ba53-19e71e36882a","resolution":{"observed_at":"2026-08-10T20:26:43.454094Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:26:43.830309Z","title":"S., Mudigere, D., Nocedal, J., Smelyanskiy, M., and Tang, P","venue":null,"work_id":"d665fab9-5ba8-4002-8fb3-f91114611d9b","year":2016},"citing_paper":{"arxiv_id":"2501.09137","last_updated":"2025-01-21T03:05:17Z","snapshot_observed_at":"2026-08-11T05:49:13.700840Z","submitted_at":"2025-01-15T20:43:36Z","title":"Gradient Descent Converges Linearly to Flatter Minima than Gradient Flow in Shallow Linear Networks","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-10T20:26:43.457845Z"},"links":{"citing_paper":"/paper/2501.09137"},"observation_digest":"sha256:eaa063e7cc58f67ee424e96e4151b16d614cdfcc6755ad817c97bbf6e2ccd4cd","observation_id":"fb4d54e9-6740-4744-9a80-a8b7f500874f","resolution":{"observed_at":"2026-08-10T20:26:43.836855Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:26:43.812939Z","title":"B., and Müller, K.-R","venue":null,"work_id":"b48cb268-8a42-4233-bb0d-820eedc9c588","year":2002},"citing_paper":{"arxiv_id":"2501.09137","last_updated":"2025-01-21T03:05:17Z","snapshot_observed_at":"2026-08-11T05:49:13.700840Z","submitted_at":"2025-01-15T20:43:36Z","title":"Gradient Descent Converges Linearly to Flatter Minima than Gradient Flow in Shallow Linear Networks","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-10T20:26:43.461625Z"},"links":{"citing_paper":"/paper/2501.09137"},"observation_digest":"sha256:8c54baf8f310bb0ebd41da21c027ccb9933be6fff94cf4afe9794f7bb1275959","observation_id":"be8bfd10-f67b-4c79-a432-26eb978412e4","resolution":{"observed_at":"2026-08-10T20:26:43.817831Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2003.02218","last_updated":"2020-03-04T17:52:48Z","snapshot_observed_at":"2026-08-08T21:43:40.840815Z","submitted_at":"2020-03-04T17:52:48Z","title":"The large learning rate phase of deep learning: the catapult mechanism","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2003.02218","snapshot_observed_at":"2026-08-10T20:26:43.466024Z","title":"The large learning rate phase of deep learning: the catapult mechanism, March 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2501.09137","last_updated":"2025-01-21T03:05:17Z","snapshot_observed_at":"2026-08-11T05:49:13.700840Z","submitted_at":"2025-01-15T20:43:36Z","title":"Gradient Descent Converges Linearly to Flatter Minima than Gradient Flow in Shallow Linear Networks","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-10T20:26:43.466024Z"},"links":{"cited_paper":"/paper/2003.02218","citing_paper":"/paper/2501.09137"},"observation_digest":"sha256:8bd222d63ab6a98151a830334bdd2f1514245df584bc233fbf946de0d6611639","observation_id":"7f64b7da-712d-495f-8161-69a21ad71df1","resolution":{"observed_at":"2026-08-10T20:26:43.466024Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:26:43.794817Z","title":"Towards explaining the regularization effect of initial large learning rate in training neural networks","venue":null,"work_id":"829e8861-25be-4089-a599-5b7d84d0e78d","year":2019},"citing_paper":{"arxiv_id":"2501.09137","last_updated":"2025-01-21T03:05:17Z","snapshot_observed_at":"2026-08-11T05:49:13.700840Z","submitted_at":"2025-01-15T20:43:36Z","title":"Gradient Descent Converges Linearly to Flatter Minima than Gradient Flow in Shallow Linear Networks","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-10T20:26:43.470884Z"},"links":{"citing_paper":"/paper/2501.09137"},"observation_digest":"sha256:efc3be3ae4dc64c3c0125ebe444f83e9710218e5685166b43ffc9c177bac3012","observation_id":"9d8049ef-d133-44ef-920f-a71bdfe7ac22","resolution":{"observed_at":"2026-08-10T20:26:43.802916Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:26:43.778881Z","title":"M., Rauhut, H., and Terstiege, U","venue":null,"work_id":"9c5aad35-f9f1-48ce-89c4-aef638fd87cb","year":2024},"citing_paper":{"arxiv_id":"2501.09137","last_updated":"2025-01-21T03:05:17Z","snapshot_observed_at":"2026-08-11T05:49:13.700840Z","submitted_at":"2025-01-15T20:43:36Z","title":"Gradient Descent Converges Linearly to Flatter Minima than Gradient Flow in Shallow Linear Networks","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-10T20:26:43.475099Z"},"links":{"citing_paper":"/paper/2501.09137"},"observation_digest":"sha256:66b6e4e9ba52104724eb2565a4448bbbdcaa1a39198457bbfb0aa48859a5bc39","observation_id":"4181a5dd-1f4d-4c31-a8c4-174f953c33b4","resolution":{"observed_at":"2026-08-10T20:26:43.784615Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:26:43.762191Z","title":"The Effect of Network Width on Stochastic Gradient Descent and Generalization : an Empirical Study","venue":null,"work_id":"50769dfc-ebe6-4dfe-9fce-a5d0d11ac325","year":2019},"citing_paper":{"arxiv_id":"2501.09137","last_updated":"2025-01-21T03:05:17Z","snapshot_observed_at":"2026-08-11T05:49:13.700840Z","submitted_at":"2025-01-15T20:43:36Z","title":"Gradient Descent Converges Linearly to Flatter Minima than Gradient Flow in Shallow Linear Networks","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-10T20:26:43.480674Z"},"links":{"citing_paper":"/paper/2501.09137"},"observation_digest":"sha256:8692dd01e550ed2ead5f8422aa38b79c3052344709f9a1d8caa6c509769addf5","observation_id":"b2a8ab38-6d47-4557-8a67-bc6e331773df","resolution":{"observed_at":"2026-08-10T20:26:43.767367Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:26:43.748833Z","title":null,"venue":null,"work_id":"5a015812-89f6-4511-99be-5a17af6cedc3","year":1963},"citing_paper":{"arxiv_id":"2501.09137","last_updated":"2025-01-21T03:05:17Z","snapshot_observed_at":"2026-08-11T05:49:13.700840Z","submitted_at":"2025-01-15T20:43:36Z","title":"Gradient Descent Converges Linearly to Flatter Minima than Gradient Flow in Shallow Linear Networks","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-10T20:26:43.485432Z"},"links":{"citing_paper":"/paper/2501.09137"},"observation_digest":"sha256:8eeebe7aa7b10ba19cd8d779e2f6fff5e1fc919f560114b02a5ca7dd8ff2367f","observation_id":"ca90ba2d-03ac-4964-967f-6c2b1c79ed06","resolution":{"observed_at":"2026-08-10T20:26:43.752819Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1312.6120","last_updated":"2014-02-19T17:26:57Z","snapshot_observed_at":"2026-08-10T15:55:45.411258Z","submitted_at":"2013-12-20T20:24:00Z","title":"Exact solutions to the nonlinear dynamics of learning in deep linear neural networks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1312.6120","snapshot_observed_at":"2026-08-10T20:26:43.492055Z","title":"M., McClelland, J","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2501.09137","last_updated":"2025-01-21T03:05:17Z","snapshot_observed_at":"2026-08-11T05:49:13.700840Z","submitted_at":"2025-01-15T20:43:36Z","title":"Gradient Descent Converges Linearly to Flatter Minima than Gradient Flow in Shallow Linear Networks","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-10T20:26:43.492055Z"},"links":{"cited_paper":"/paper/1312.6120","citing_paper":"/paper/2501.09137"},"observation_digest":"sha256:c71445a0ae4cb4c146dab3a9e057bb580f5a9fb4df39d6fb6cdb535d26607bd9","observation_id":"2dd077db-079a-4f7d-bebe-0efc2e47a981","resolution":{"observed_at":"2026-08-10T20:26:43.492055Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1710.06451","last_updated":"2018-02-14T19:42:20Z","snapshot_observed_at":"2026-08-10T06:38:26.267516Z","submitted_at":"2017-10-17T18:08:04Z","title":"A Bayesian Perspective on Generalization and Stochastic Gradient Descent","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1710.06451","snapshot_observed_at":"2026-08-10T20:26:43.497375Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2501.09137","last_updated":"2025-01-21T03:05:17Z","snapshot_observed_at":"2026-08-11T05:49:13.700840Z","submitted_at":"2025-01-15T20:43:36Z","title":"Gradient Descent Converges Linearly to Flatter Minima than Gradient Flow in Shallow Linear Networks","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-10T20:26:43.497375Z"},"links":{"cited_paper":"/paper/1710.06451","citing_paper":"/paper/2501.09137"},"observation_digest":"sha256:867384599f16d7ab45cefdbfd1dc34f03121d674999b8dc6178888ec4ffa7c3f","observation_id":"fef5958b-0a7e-4721-b896-35d878e34995","resolution":{"observed_at":"2026-08-10T20:26:43.497375Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:26:43.731552Z","title":"D., and Vidal, R","venue":null,"work_id":"207a6901-4648-457f-993e-5b98dd472126","year":2021},"citing_paper":{"arxiv_id":"2501.09137","last_updated":"2025-01-21T03:05:17Z","snapshot_observed_at":"2026-08-11T05:49:13.700840Z","submitted_at":"2025-01-15T20:43:36Z","title":"Gradient Descent Converges Linearly to Flatter Minima than Gradient Flow in Shallow Linear Networks","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-10T20:26:43.502061Z"},"links":{"citing_paper":"/paper/2501.09137"},"observation_digest":"sha256:7512affa607123e246003e30a098ae069a04884b4596a50f05a1e37c30741d2b","observation_id":"a9308a71-ecfb-4aa5-bc96-0aba767f8dbf","resolution":{"observed_at":"2026-08-10T20:26:43.737100Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:26:43.711092Z","title":"Large Learning Rate Tames Homogeneity : Convergence and Balancing Effect","venue":null,"work_id":"8acd5b63-008d-4a08-9850-eceb1bfc04eb","year":2022},"citing_paper":{"arxiv_id":"2501.09137","last_updated":"2025-01-21T03:05:17Z","snapshot_observed_at":"2026-08-11T05:49:13.700840Z","submitted_at":"2025-01-15T20:43:36Z","title":"Gradient Descent Converges Linearly to Flatter Minima than Gradient Flow in Shallow Linear Networks","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-10T20:26:43.508487Z"},"links":{"citing_paper":"/paper/2501.09137"},"observation_digest":"sha256:cf3d3d6abfee5c3435aa304506d3fd18621663fc5e70ec1229524db1f72928bb","observation_id":"a26044fc-ce5c-4f30-bf6e-eeab91f6b694","resolution":{"observed_at":"2026-08-10T20:26:43.717299Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07085","last_updated":"2025-02-21T11:50:09Z","snapshot_observed_at":"2026-08-04T06:52:50.677485Z","submitted_at":"2024-01-13T14:21:46Z","title":"Three Mechanisms of Feature Learning in a Linear Network","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.07085","snapshot_observed_at":"2026-08-10T20:26:43.512820Z","title":"and Ziyin, L","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.09137","last_updated":"2025-01-21T03:05:17Z","snapshot_observed_at":"2026-08-11T05:49:13.700840Z","submitted_at":"2025-01-15T20:43:36Z","title":"Gradient Descent Converges Linearly to Flatter Minima than Gradient Flow in Shallow Linear Networks","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-10T20:26:43.512820Z"},"links":{"cited_paper":"/paper/2401.07085","citing_paper":"/paper/2501.09137"},"observation_digest":"sha256:1951f041450546aee8427ad6526506c7eb0dbd60d4b9dde10b236ff0ee9daed5","observation_id":"edd11cbf-48dc-408a-9fe7-0f43af6fa797","resolution":{"observed_at":"2026-08-10T20:26:43.512820Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:26:43.694871Z","title":"Linear convergence of gradient descent for finite width over-parametrized linear networks with general initialization","venue":null,"work_id":"82ce45fb-7c79-47d1-9347-98520ef680ec","year":2023},"citing_paper":{"arxiv_id":"2501.09137","last_updated":"2025-01-21T03:05:17Z","snapshot_observed_at":"2026-08-11T05:49:13.700840Z","submitted_at":"2025-01-15T20:43:36Z","title":"Gradient Descent Converges Linearly to Flatter Minima than Gradient Flow in Shallow Linear Networks","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-10T20:26:43.516613Z"},"links":{"citing_paper":"/paper/2501.09137"},"observation_digest":"sha256:2b978ac981a1378a24451f6d5255c1c67c08d4fe4202c3e7a4d9c13fb8cc5ef9","observation_id":"0952d60f-06b8-453c-b4c8-c883720e3fd0","resolution":{"observed_at":"2026-08-10T20:26:43.700426Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:26:43.521281Z","title":"@esa (Ref","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.09137","last_updated":"2025-01-21T03:05:17Z","snapshot_observed_at":"2026-08-11T05:49:13.700840Z","submitted_at":"2025-01-15T20:43:36Z","title":"Gradient Descent Converges Linearly to Flatter Minima than Gradient Flow in Shallow Linear Networks","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-10T20:26:43.521281Z"},"links":{"citing_paper":"/paper/2501.09137"},"observation_digest":"sha256:5a0251d449ff3b21ec2c673fea54fa64c71e6d8f9bcff9ff5afba531d5a54b7d","observation_id":"7aa214f3-1596-4b92-a4ec-3480c1750796","resolution":{"observed_at":"2026-08-10T20:26:43.521281Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:26:43.525592Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.09137","last_updated":"2025-01-21T03:05:17Z","snapshot_observed_at":"2026-08-11T05:49:13.700840Z","submitted_at":"2025-01-15T20:43:36Z","title":"Gradient Descent Converges Linearly to Flatter Minima than Gradient Flow in Shallow Linear Networks","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-10T20:26:43.525592Z"},"links":{"citing_paper":"/paper/2501.09137"},"observation_digest":"sha256:fe0199e9b22c4c96e15ca33b89af574b549261a15edbb630abb18f6e09f545e7","observation_id":"5e389e42-4f56-4eaa-bd43-7f8f4921c8b6","resolution":{"observed_at":"2026-08-10T20:26:43.525592Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:26:43.659219Z","title":",# (7),01444 '9=82<.342C 2! !22222222222222222222222222222222222222222222222222","venue":null,"work_id":"133a5269-aca1-448a-9ccd-4ea009d0d946","year":null},"citing_paper":{"arxiv_id":"2501.09137","last_updated":"2025-01-21T03:05:17Z","snapshot_observed_at":"2026-08-11T05:49:13.700840Z","submitted_at":"2025-01-15T20:43:36Z","title":"Gradient Descent Converges Linearly to Flatter Minima than Gradient Flow in Shallow Linear Networks","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-10T20:26:43.530125Z"},"links":{"citing_paper":"/paper/2501.09137"},"observation_digest":"sha256:5cd143b01a6ae358d0d334d330750271ceddd7f8f2b757ddf7c1577e710dfa0a","observation_id":"b494e2f5-df1c-419d-8b51-6f59a2a6230a","resolution":{"observed_at":"2026-08-10T20:26:43.666612Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2501.09137","last_updated":"2025-01-21T03:05:17Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-11T05:49:13.700840Z","submitted_at":"2025-01-15T20:43:36Z","title":"Gradient Descent Converges Linearly to Flatter Minima than Gradient Flow in Shallow Linear Networks"},"reference_resolution":{"displayed":26,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":11,"verified_exact":0,"verified_fuzzy":15},"total_outbound_references":26},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 26 of 26 outbound references and 1 inbound Pith citation observation for arXiv:2501.09137."}