{"as_of":"2026-08-15T19:36:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c46fd8353b2507ff2d1db88d0bf6dacc0e9cbaecfeb1df92d12944cc7340c906","coverage":[{"denominator":55,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":55,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T18:38:57.105705Z","state":"measured"},{"denominator":56,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":56,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-30T07:05:09.767632Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-06-30T07:14:21.720119Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"cited_work":{"arxiv_id":"2502.05668","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.05668","snapshot_observed_at":"2026-06-30T07:14:21.720119Z","title":"arXiv preprint arXiv:2502.05668 , year=","venue":null,"work_id":"0b983cd1-0668-4146-bb78-e73356248610","year":null},"citing_paper":{"arxiv_id":"2606.30559","last_updated":"2026-06-29T16:55:45Z","snapshot_observed_at":"2026-08-01T14:52:02.260041Z","submitted_at":"2026-06-29T16:55:45Z","title":"Convergence of Continual Learning in Homogeneous Deep Networks","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-06-30T07:05:09.767632Z"},"links":{"cited_paper":"/paper/2502.05668","citing_paper":"/paper/2606.30559"},"observation_digest":"sha256:1cfeac29387037854836c96b7bb1787267235414f8dfdcc49f2e88a8a78f27c4","observation_id":"deb5d617-a54b-408f-8faa-fc0a7131ccc2","resolution":{"observed_at":"2026-06-30T07:14:21.722003Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2502.05668/citation-record","integrity":"/paper/2502.05668/integrity","json":"/paper/2502.05668/citation-record.json","paper":"/paper/2502.05668"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:56.159190Z","title":null,"venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.159190Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:c81133ce0af68a9160f0d2a662e154450542b3efe3ea9e27e55c41828cbe5123","observation_id":"7fbd825b-9009-4509-a187-5b9ecde6c088","resolution":{"observed_at":"2026-08-08T18:38:56.159190Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:56.168317Z","title":"Reconciling modern machine-learning practice and the classical bias--variance trade-off","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.168317Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:3b3071a6c2432724478d6969d0a3c4c23513ab2054aca3f31d211f64cae6c478","observation_id":"5974750c-0dd6-4bf3-b8e9-d49bb088d508","resolution":{"observed_at":"2026-08-08T18:38:56.168317Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:59.032117Z","title":"Dynamics of stochastic approximation algorithms","venue":null,"work_id":"92441a6c-3d1c-4375-9afb-6b13ce7491a2","year":2006},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.204755Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:a57bb79307cc4b86791376c58b5babfa733fabffb8288fa81fbf7c13aeab9c3a","observation_id":"e8d0c3ae-210f-4caa-b56c-7cb1896353cd","resolution":{"observed_at":"2026-08-08T18:38:59.036744Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:59.017988Z","title":"Stochastic approximations and differential inclusions","venue":null,"work_id":"88fb3462-e65a-4e73-82cf-efda853f7635","year":2005},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.254753Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:7d92bc5b7c895a31d670769c5479fc2f5af7751512e30b44183ca1ea9f5955bb","observation_id":"599f30be-afa9-40c7-a098-56d7cec4ae48","resolution":{"observed_at":"2026-08-08T18:38:59.022558Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:59.003284Z","title":"Semianalytic and subanalytic sets","venue":null,"work_id":"0f6aab93-a7ae-4c2c-b61e-1bd968eaf539","year":1988},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.267684Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:600552eec4d963ef82630b2469600cc7892897e4c99b7372d69916e922192558","observation_id":"e72189df-7430-4d82-b0e7-067cc535f5e4","resolution":{"observed_at":"2026-08-08T18:38:59.007741Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:58.987763Z","title":"Bolte, A","venue":null,"work_id":"82730fe4-7122-4a46-845b-77191d7a2c8a","year":2007},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.272736Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:2bb34b7a0e2cd85ff546eaa51748cfbd7b63b866a741f721d88ac5c9248ee147","observation_id":"1744fd61-b6b9-422b-a5f8-1391d92959a4","resolution":{"observed_at":"2026-08-08T18:38:58.992923Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:58.970428Z","title":"Conservative set valued fields, automatic differentiation, stochastic gradient methods and deep learning","venue":null,"work_id":"3db82b46-f94e-4a10-b806-aac020711def","year":2021},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.277712Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:5f9325d23ab9a3d54c26545355ced6267a938b19a690b4d1b234ffa41fcb4029","observation_id":"8939745b-0806-4ad0-9121-ff69cb25b8c3","resolution":{"observed_at":"2026-08-08T18:38:58.976775Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:58.925417Z","title":"Subgradient sampling for nonsmooth nonconvex minimization","venue":null,"work_id":"00029491-2dd6-4df7-91f9-c0155e50fe97","year":2023},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.283711Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:b81923b6bd075cb6f0a13e176d83c478c0dfba2fef7216fde39e30407f0d393b","observation_id":"248dc712-762b-4633-ade3-c8de2e3ae374","resolution":{"observed_at":"2026-08-08T18:38:58.959502Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:58.756490Z","title":"Stochastic approximation: a dynamical systems viewpoint, volume 9","venue":null,"work_id":"18927d29-0113-43ba-8db5-afcd0cbb4c28","year":2008},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.288774Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:98f7c892747b2fb40eb3a38d9af80dd39e6847eedd45193f64453bfc7c0bc966","observation_id":"385ae350-d150-4320-844b-80957b117c4a","resolution":{"observed_at":"2026-08-08T18:38:58.809984Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:58.642083Z","title":"The ode method for convergence of stochastic approximation and reinforcement learning","venue":null,"work_id":"97b35a1e-f5c5-47b5-a6d3-409cd5a96642","year":2000},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.294148Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:6f1f6ed50feac55f3b4b59ef094a65d938ea4a237f7b035a3f02a717611046d2","observation_id":"67b47f7f-8ba1-4837-80be-63cd3d6a0ffb","resolution":{"observed_at":"2026-08-08T18:38:58.711709Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:56.304944Z","title":"An introduction to optimization on smooth manifolds","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.304944Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:4d8f89304aa7bd7c85b73936676e29cc68f93c3372c83b8ecff68477f9f7bc6c","observation_id":"fe38cc36-f071-4e8b-88a9-ce8d4d06dc1b","resolution":{"observed_at":"2026-08-08T18:38:56.304944Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:58.600932Z","title":"Large stepsize gradient descent for non-homogeneous two-layer networks: Margin improvement and fast optimization","venue":null,"work_id":"af5f8e8d-0f0c-4d15-af50-9e4ac8c3c091","year":2024},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.311484Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:11a26bb8af3d8e193d5b8d7f1364223808b8d293711a8e83bf1a8d940479fd76","observation_id":"f3966ade-874b-4499-b74f-cfd57faf6315","resolution":{"observed_at":"2026-08-08T18:38:58.605754Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:56.315814Z","title":"Implicit bias of gradient descent for wide two-layer neural networks trained with the logistic loss","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.315814Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:55a5629dba9f0e85e71d864cb30c1961693fc00d81e0e59854e17717eed20100","observation_id":"5c94012c-62d7-46ea-94fb-deae84c99267","resolution":{"observed_at":"2026-08-08T18:38:56.315814Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:58.578565Z","title":"Nonsmooth analysis and control theory, volume 178","venue":null,"work_id":"5d5e93af-4287-4f3a-96fa-ee21d4a89cb6","year":1998},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.320365Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:1123909ecd256de1420af3ac1482c37ba70a80f673f1fe2086b4fac7ff99d0df","observation_id":"2ac8ba6a-21da-4f1d-bc8c-145c61d206a9","resolution":{"observed_at":"2026-08-08T18:38:58.582629Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:58.564990Z","title":"An introduction to o-minimal geometry","venue":null,"work_id":"69ba1fb6-8f45-42ef-984b-9b85a219bf74","year":2000},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.340621Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:8d8813bf01319215efc2507bfc06931d7e932811a4451277a3bf51d4ec41ef5e","observation_id":"7a6a4062-a337-475f-93a1-03269016620d","resolution":{"observed_at":"2026-08-08T18:38:58.569297Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:58.550958Z","title":"Stochastic subgradient method converges on tame functions","venue":null,"work_id":"64128b2e-0e17-4c5f-b1b3-24266e871b1c","year":2020},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.345930Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:fece28dae189c0da6ff511f16fdbf6078c01f408cd94bf5a75b68f7689b33a3b","observation_id":"a38c5ba8-b4b5-4bbc-bbe2-578bce8d9ad8","resolution":{"observed_at":"2026-08-08T18:38:58.556177Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:58.537254Z","title":"Curves of descent","venue":null,"work_id":"1c5b3fd3-5d4b-4bf2-9c3b-ca4a096d0b59","year":2015},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.350709Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:68e745271788f89c70b55cf467dd4c71d3117768a5746df67f25520d0227693b","observation_id":"db68b9eb-f9c8-4dcb-a050-99801f471346","resolution":{"observed_at":"2026-08-08T18:38:58.541356Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:58.522961Z","title":"Algorithmic regularization in learning deep homogeneous models: Layers are automatically balanced","venue":null,"work_id":"54079f9c-753c-4604-8aee-6cb68f61b4f5","year":2018},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.356730Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:bdd13332c67580808c44463075ba99c03042ba54639901868dca4fe2f6d1add6","observation_id":"d0ad4bc6-fa09-4f43-a6b5-2de151bcff96","resolution":{"observed_at":"2026-08-08T18:38:58.527567Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:58.508177Z","title":"Stochastic methods for composite and weakly convex optimization problems","venue":null,"work_id":"29cffd10-b50c-4d7d-b07c-e761cb8b3d9e","year":2018},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.361924Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:a2eebc8286a0cfb14ce5a66052b70c39d9c733a84187871e5b06d00b547c040d","observation_id":"9a389864-f43c-45ee-a055-01dd65abd8fe","resolution":{"observed_at":"2026-08-08T18:38:58.512667Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:58.494299Z","title":"The little book of deep learning","venue":null,"work_id":"b4d87f2b-61e5-4410-b9c5-a6ba46df9575","year":2023},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.374816Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:2009c66cf4e0c0dbf3e726b60291f3323fe11642729263fe326fa25f63c4c11f","observation_id":"ef1cc023-19d9-4b1b-8c95-beb4d6c0832d","resolution":{"observed_at":"2026-08-08T18:38:58.499220Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:58.479447Z","title":"Complements of subanalytic sets and existential formulas for analytic functions","venue":null,"work_id":"c23f9af8-a7bc-4fb9-9feb-59e6a72ef718","year":1996},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.379994Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:0eb961c1ab5d52cd2bd5852add0473aedb0893f411ecd6e58fc9d73b5458e752","observation_id":"b0cb0a3d-170c-430a-a736-44a925b08576","resolution":{"observed_at":"2026-08-08T18:38:58.484778Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:58.464003Z","title":"Projections of semi-analytic sets","venue":null,"work_id":"5ed282d9-3add-4e7f-8d8c-61f6bed40592","year":1968},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.384476Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:bc60b33fe1bb9f093f352bc206fa07fe57eb7d22e687a456a8c37c48b19ca511","observation_id":"1e93a0a1-2866-43a0-a7d9-176bd12d7fe6","resolution":{"observed_at":"2026-08-08T18:38:58.468335Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:58.449388Z","title":"Deep learning, volume 196","venue":null,"work_id":"4793e28f-3b21-45e6-9ba4-cdf979f00fc0","year":2016},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.389061Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:cfc215213936ff17e047eb6abdc1cc9b56b3f2c29e78ab3a821ecb0b31080ad9","observation_id":"2da008c0-3f64-4ec3-a2fa-7024081c2154","resolution":{"observed_at":"2026-08-08T18:38:58.454554Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:58.249523Z","title":"Lee, Daniel Soudry, and Nathan Srebro","venue":null,"work_id":"7d949b55-ee90-4970-a3a8-3c10d6b05593","year":2018},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.393550Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:5b0d27c61864538dfb26fe3f1f20f0703c5af2f27043a4eee7d755e3d5e535d9","observation_id":"48bae627-5ca6-4aa2-9a01-b72070a9a39f","resolution":{"observed_at":"2026-08-08T18:38:58.397752Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:58.110198Z","title":"Implicit bias of gradient descent on linear convolutional networks","venue":null,"work_id":"851ac3e5-9426-45ba-b602-b540759731fa","year":2018},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.398283Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:31d2a5364da0bacc2ded378ef7abfd7a184d50d00363d556ca8f8f35c944c336","observation_id":"0d88e792-9acd-4924-a275-6afd390abce2","resolution":{"observed_at":"2026-08-08T18:38:58.166220Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:58.061128Z","title":"An invitation to tame optimization","venue":null,"work_id":"7fbdb9f4-e637-4a9e-901b-643b8cec561c","year":1917},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.402847Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:5e49ef062bea4aaa0033f7fa85c3ab3c67750a9e0a234c67476effcbf50fd9e6","observation_id":"a6e19c96-d9f1-4f7c-b6de-5cd1dcfed0f7","resolution":{"observed_at":"2026-08-08T18:38:58.081772Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1810.02032","last_updated":"2019-02-24T10:28:05Z","snapshot_observed_at":"2026-08-15T12:41:19.752363Z","submitted_at":"2018-10-04T02:48:41Z","title":"Gradient descent aligns the layers of deep linear networks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1810.02032","snapshot_observed_at":"2026-08-08T18:38:56.407470Z","title":"Gradient descent aligns the layers of deep linear networks","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.407470Z"},"links":{"cited_paper":"/paper/1810.02032","citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:2c7f979ace35cbc9e05c96a3d849815f84865230d5d8fc20431a299b6f0cffb1","observation_id":"81ba1b4d-0540-4949-8afb-97aac81c8c58","resolution":{"observed_at":"2026-08-08T18:38:56.407470Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1803.07300","last_updated":"2019-06-08T13:57:05Z","snapshot_observed_at":"2026-08-14T19:34:29.603750Z","submitted_at":"2018-03-20T08:47:27Z","title":"Risk and parameter convergence of logistic regression","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1803.07300","snapshot_observed_at":"2026-08-08T18:38:56.442270Z","title":"Risk and parameter convergence of logistic regression","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.442270Z"},"links":{"cited_paper":"/paper/1803.07300","citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:f922af5751deb7f5b5501f538f5731c7807fb6e97a98dac8386b864abf6d86a4","observation_id":"c89ad522-114e-4e22-997d-3513847096d5","resolution":{"observed_at":"2026-08-08T18:38:56.442270Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:56.534824Z","title":"Directional convergence and alignment in deep learning","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.534824Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:5773be2ad45b98ff889bd9c3e4ae6497519a8a1ba67f356ca919d6e51a676ebf","observation_id":"1bdb5dee-dad4-4377-8ec1-6d54f5f0f7f1","resolution":{"observed_at":"2026-08-08T18:38:56.534824Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:58.017243Z","title":"Global stability of first-order methods for coercive tame functions","venue":null,"work_id":"b19b3462-f6f7-414e-ad6b-a216b65458e4","year":2024},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.610645Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:686bb7ab6174a80ddaddb52e918e0dfbbf1e091bde746bc0b72708373dc88e81","observation_id":"1700a4a0-4005-44a0-a4d6-513b58a60ecb","resolution":{"observed_at":"2026-08-08T18:38:58.022282Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.03820","last_updated":"2023-02-17T00:30:31Z","snapshot_observed_at":"2026-08-13T14:10:14.670317Z","submitted_at":"2022-10-07T21:14:09Z","title":"The Asymmetric Maximum Margin Bias of Quasi-Homogeneous Neural Networks","version":2},"cited_work":{"arxiv_id":"2210.03820","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.03820","snapshot_observed_at":"2026-08-08T18:38:57.502688Z","title":"The Asymmetric Maximum Margin Bias of Quasi-Homogeneous Neural Networks","venue":"cs.LG","work_id":"51364487-ccd5-47e2-bcd4-7508d8340eb5","year":2022},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.677853Z"},"links":{"cited_paper":"/paper/2210.03820","citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:72a66ed57405b58f8e56edb2f5a60a78254858dfbd371a899fcea49b0178ea3e","observation_id":"29b04d98-57a6-48a7-ac36-3a128c0802d6","resolution":{"observed_at":"2026-08-08T18:38:57.508781Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:56.728950Z","title":"An Introduction to Differential Manifolds","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.728950Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:5d3c9e19d5655137073c5eeea322b5fd2c33671f2b319ece784cc6ac0551412f","observation_id":"5b3c22b2-59c4-45c5-a512-5c9754de0259","resolution":{"observed_at":"2026-08-08T18:38:56.728950Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:58.002017Z","title":"Nonsmooth nonconvex stochastic heavy ball","venue":null,"work_id":"39d0e957-58b5-481b-bc11-2cb95aab4523","year":2024},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.771119Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:35819b85c34c2cb9559f9d4189edfaaccffa0aa5d1b2cffb8fef18e44dcdfb03","observation_id":"123533da-3681-4ccb-ac40-3f2d2ab1aabd","resolution":{"observed_at":"2026-08-08T18:38:58.006814Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2201.11968","last_updated":"2022-04-26T03:14:57Z","snapshot_observed_at":"2026-08-15T19:32:25.947308Z","submitted_at":"2022-01-28T07:31:19Z","title":"Training invariances and the low-rank phenomenon: beyond linear networks","version":2},"cited_work":{"arxiv_id":"2201.11968","doi":null,"metadata_source":"pith","pith_arxiv_id":"2201.11968","snapshot_observed_at":"2026-08-08T18:38:57.481119Z","title":"Training invariances and the low-rank phenomenon: beyond linear networks","venue":"cs.LG","work_id":"dd9186a8-b9bb-4ca8-8eb6-34eb3e081c2d","year":2022},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.838459Z"},"links":{"cited_paper":"/paper/2201.11968","citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:9adcf872402e30e56906b1e2731bc9025d4c88631a6d999641c7c553a46860d6","observation_id":"d654da66-7c8a-437d-a93c-2fb12654215e","resolution":{"observed_at":"2026-08-08T18:38:57.486338Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:57.986938Z","title":"Gradient descent maximizes the margin of homogeneous neural networks","venue":null,"work_id":"1e110105-417b-4635-9f2f-e009ae30077e","year":2020},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.843765Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:8c45e3308dd05d4421fe8872b0b1e779c4dc8c2186095b6b9caf8f4ca7b2d048","observation_id":"3a8c4171-8fbe-4620-96b2-495caac1ff02","resolution":{"observed_at":"2026-08-08T18:38:57.991657Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1805.01916","last_updated":"2018-05-04T19:13:56Z","snapshot_observed_at":"2026-08-14T19:18:39.904879Z","submitted_at":"2018-05-04T19:13:56Z","title":"Analysis of nonsmooth stochastic approximation: the differential inclusion approach","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1805.01916","snapshot_observed_at":"2026-08-08T18:38:56.850571Z","title":"Analysis of nonsmooth stochastic approximation: the differential inclusion approach","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.850571Z"},"links":{"cited_paper":"/paper/1805.01916","citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:dc4a6b35bb81d58b6932e8f5bd4a51b999725f0856c56411b73728d8d144dc5e","observation_id":"9210eff0-c4be-4ea7-aafc-ed763dc729b8","resolution":{"observed_at":"2026-08-08T18:38:56.850571Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:57.971935Z","title":"Lexicographic and depth-sensitive margins in homogeneous and non-homogeneous deep models","venue":null,"work_id":"f419db6f-e23b-44ee-a431-b7d779b5a554","year":2019},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.855892Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:5da6d92030d2590f89f78dd4f1e9b6da27e02ee6dc6b701b572d985673e1bb94","observation_id":"dac1f880-7043-4eb2-9732-33efd0f70bfe","resolution":{"observed_at":"2026-08-08T18:38:57.976710Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:57.956942Z","title":"Convergence of gradient descent on separable data","venue":null,"work_id":"63a43641-98c7-4c7f-8958-54eb33bb86f1","year":2019},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.860778Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:e1ddb16ddefd0218ec7aa696f1133ce49477b0031dfbd42fbd5a7d9aeef794cd","observation_id":"3a689346-8e30-4914-b834-2092040b2bd4","resolution":{"observed_at":"2026-08-08T18:38:57.961752Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:57.941930Z","title":"Stochastic gradient descent on separable data: Exact convergence with a fixed learning rate","venue":null,"work_id":"322542db-e4cc-40dc-8707-61b4ba84cc95","year":2019},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.867822Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:03c0b64912a0705577090200e93d88f11b5a228e014b83d9a3a278260cc67ea9","observation_id":"8a7d30e8-3b4b-4801-88d8-6d2c5c0b77e8","resolution":{"observed_at":"2026-08-08T18:38:57.947869Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1412.6614","last_updated":"2015-04-16T18:48:31Z","snapshot_observed_at":"2026-08-14T23:06:39.936635Z","submitted_at":"2014-12-20T06:52:25Z","title":"In Search of the Real Inductive Bias: On the Role of Implicit Regularization in Deep Learning","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1412.6614","snapshot_observed_at":"2026-08-08T18:38:56.872522Z","title":"In search of the real inductive bias: On the role of implicit regularization in deep learning","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.872522Z"},"links":{"cited_paper":"/paper/1412.6614","citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:3ed9adbdc987773e701636e5ba06042b9624dcbfa9fa1b6dac0f78754045b44f","observation_id":"998c7d05-dd68-4df1-85d5-7f45bac7a697","resolution":{"observed_at":"2026-08-08T18:38:56.872522Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:56.884296Z","title":"Automatic differentiation in pytorch","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.884296Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:5465490593ed6ab4a13133eb653263a58a996a111c43071f2a94f71b099a445a","observation_id":"e6674e19-90c5-41dd-9799-5c63176751c3","resolution":{"observed_at":"2026-08-08T18:38:56.884296Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:57.908316Z","title":"A generalization of the borkar-meyn theorem for stochastic recursive inclusions","venue":null,"work_id":"f48c3550-81c3-4d76-8cc4-530949abc10b","year":2017},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.894516Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:66310efaa698659dfaf5fdac10393cce458429e69e9bae32d4cd3d1a93247be9","observation_id":"7b3d92c4-0ff8-4510-899b-dc143eefb9c7","resolution":{"observed_at":"2026-08-08T18:38:57.913688Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"item/1183504","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:57.420349Z","title":"The measure of the critical values of differentiable maps","venue":null,"work_id":"06f02422-4d9a-46b1-8e49-efda0d60b9bf","year":1942},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.902062Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:8df27799a3b80e624c220a1a978b9b752558725ef2dfdb7d646847d869a1ec7f","observation_id":"aad8fb48-f1af-4f9e-bd80-c6ba042d94d4","resolution":{"observed_at":"2026-08-08T18:38:57.431522Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:56.906980Z","title":"The implicit bias of gradient descent on separable data","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.906980Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:9fd5e4a40060fa5dcef43ac9444a8b9e7fc3bfa86d5fc7015e2b2fc7576d8e8e","observation_id":"df8e8f68-f04a-4122-bdcf-7644a6b32d27","resolution":{"observed_at":"2026-08-08T18:38:56.906980Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:56.918027Z","title":"A decision method for elementary algebra and geometry","venue":null,"work_id":null,"year":1951},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.918027Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:6cee254c0505be0dffe54ccce8d68300e972401a901a0b8118543968eaf1025a","observation_id":"4f0c7596-c088-4163-aa9e-633a1706da45","resolution":{"observed_at":"2026-08-08T18:38:56.918027Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:57.873494Z","title":"Tame topology and o-minimal structures, volume 248","venue":null,"work_id":"7a1c5cac-8a4c-41cd-a05f-c06c0bc79170","year":1998},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.923906Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:260a5b38e2e398bcaa4824de03b8dbb07d1c1eaa5163cb3dcecbd0f4479aa170","observation_id":"3e8e4315-7065-49f7-8c03-9ecd2d3c5c55","resolution":{"observed_at":"2026-08-08T18:38:57.878862Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:57.856882Z","title":"Geometric categories and o-minimal structures","venue":null,"work_id":"6790f509-f928-4420-a221-155097cd3268","year":1996},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.929138Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:34e4966ac5ef1df9524be0f5e7825f7ad9a2687a994d8564cfce252d36019682","observation_id":"99b34755-0249-4fee-9118-e63f978e1fcb","resolution":{"observed_at":"2026-08-08T18:38:57.861574Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:57.815621Z","title":"The elementary theory of restricted analytic fields with exponentiation","venue":null,"work_id":"8898241e-cd83-4c6a-ae4e-1a35c08ac01f","year":1994},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.934461Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:a3a42fe8b03f0e34bcec657612f383c65ff9b79c3d6ab3f26cce7011ee4c6690","observation_id":"bb0c2ed5-7f49-40ff-9392-b8510455cb3f","resolution":{"observed_at":"2026-08-08T18:38:57.846335Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:57.636351Z","title":"Statistical learning theory","venue":null,"work_id":"dc47bb9a-1dba-472a-94c5-2b0ee36f3364","year":1998},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.939350Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:b05f0e759b93d35f1cfb3d5802fd67092579e67f35917959d7d9ea2eaf4679ca","observation_id":"98c4737c-70c0-4c61-b57d-23f1f2f5f51e","resolution":{"observed_at":"2026-08-08T18:38:57.736656Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:56.946462Z","title":"On the implicit bias in deep-learning algorithms","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.946462Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:54ec219066bc80096b707d03236a38cd0d145ace3dc0aff6ff17682b3bf7ea2b","observation_id":"dfc36c98-a3b7-473e-87c0-e2624bf85633","resolution":{"observed_at":"2026-08-08T18:38:56.946462Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:57.584448Z","title":"On margin maximization in linear and relu networks","venue":null,"work_id":"5824edc0-036f-423e-8b9f-151aa22e406a","year":2022},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.953033Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:afe265830bd0590252f9e1dd1e0042ff9759af079d66e6b61965eccdf01f84c3","observation_id":"1f2c7d03-ad75-40fc-b2fe-a9a3a226391b","resolution":{"observed_at":"2026-08-08T18:38:57.589078Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:57.569886Z","title":"The implicit bias for adaptive optimization algorithms on homogeneous neural networks","venue":null,"work_id":"a922403f-a60a-4b22-a468-251fd804f1f1","year":2021},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.958243Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:aae9dcc9600fb36f8d7c6f6a7d4ade0c1d68d3321925c9270e5679cf49418ca5","observation_id":"93f113f2-3cda-4fb2-8414-80abb7aebd87","resolution":{"observed_at":"2026-08-08T18:38:57.574800Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:57.554741Z","title":"Model completeness results for expansions of the ordered field of real numbers by restricted pfaffian functions and the exponential function","venue":null,"work_id":"d2c65819-5ae0-45b2-9aa0-886414fcea66","year":1996},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.963459Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:5ac4f9dc232226230d204e039684c52d84c59a567c4a9c0e856e7c604d6ac150","observation_id":"14db0bdc-5cbd-42c0-a0b7-af8dfce8fc0c","resolution":{"observed_at":"2026-08-08T18:38:57.559818Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.02501","last_updated":"2021-09-10T05:33:27Z","snapshot_observed_at":"2026-08-08T09:41:56.134951Z","submitted_at":"2020-10-06T06:08:35Z","title":"A Unifying View on Implicit Bias in Training Linear Neural Networks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.02501","snapshot_observed_at":"2026-08-08T18:38:56.984487Z","title":"A unifying view on implicit bias in training linear neural networks","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:56.984487Z"},"links":{"cited_paper":"/paper/2010.02501","citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:4509f57a43a075774d90a339c85a79d386d8734598225ef1de2509f73fb70a92","observation_id":"6804c035-5aaf-44cb-a657-2d55be3cbbc6","resolution":{"observed_at":"2026-08-08T18:38:56.984487Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T18:38:57.105705Z","title":"Understanding deep learning (still) requires rethinking generalization","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks","version":3},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-08T18:38:57.105705Z"},"links":{"citing_paper":"/paper/2502.05668"},"observation_digest":"sha256:0375982456313311e5f935960d02d5f6d1dda0dc16b24a4bf60486f2072b53b7","observation_id":"3e3b6691-8f37-4cc8-a5e7-8800f3dc51fb","resolution":{"observed_at":"2026-08-08T18:38:57.105705Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2502.05668","last_updated":"2025-07-17T11:44:07Z","latest_version":3,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-15T17:21:13.334439Z","submitted_at":"2025-02-08T19:09:16Z","title":"The late-stage training dynamics of (stochastic) subgradient descent on homogeneous neural networks"},"reference_resolution":{"displayed":55,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":16,"verified_exact":3,"verified_fuzzy":36},"total_outbound_references":55},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 55 of 55 outbound references and 1 inbound Pith citation observation for arXiv:2502.05668."}