{"as_of":"2026-08-16T20:33:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4bda97241006120105bef83efa3a8d1a60f73a3bd744fc81d25b5eaea06e5bb7","coverage":[{"denominator":94,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":94,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T10:35:07.689364Z","state":"measured"},{"denominator":99,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":99,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-16T06:30:59.297886+00:00","state":"measured"},{"denominator":5,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":5,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T06:04:02.505287Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-06-30T19:55:01.085125Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.18208","snapshot_observed_at":"2026-08-07T06:04:02.505287Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.06584","last_updated":"2025-06-06T23:32:38Z","snapshot_observed_at":"2026-08-13T10:33:16.974743Z","submitted_at":"2025-06-06T23:32:38Z","title":"Global Convergence of Gradient EM for Over-Parameterized Gaussian Mixtures","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-07T06:04:02.505287Z"},"links":{"cited_paper":"/paper/2504.18208","citing_paper":"/paper/2506.06584"},"observation_digest":"sha256:e8a9b2132881104b661e0686d2c89ebb8cb2da2fa4c0de59d744b890fb6e9daa","observation_id":"2c071eb5-c3ce-4474-adad-e0b67ce530f9","resolution":{"observed_at":"2026-08-07T06:04:02.505287Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"cited_work":{"arxiv_id":"2504.18208","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.18208","snapshot_observed_at":"2026-06-30T19:55:01.085125Z","title":"Ultra-fast fea- ture learning for the training of two-layer neural net- works in the two-timescale regime.arXiv preprint arXiv:2504.18208","venue":null,"work_id":"74802858-d44a-4c3c-9f04-b14627b4cfb3","year":null},"citing_paper":{"arxiv_id":"2510.04606","last_updated":"2026-05-08T12:28:24Z","snapshot_observed_at":"2026-08-13T14:56:58.340356Z","submitted_at":"2025-10-06T09:14:39Z","title":"Closed-Form Last Layer Optimization","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-18T09:58:50.815850Z"},"links":{"cited_paper":"/paper/2504.18208","citing_paper":"/paper/2510.04606"},"observation_digest":"sha256:5b1ce96bbd9d462ff770734b3f8a9c8e3dda3cf2a93709ed59e12af78d6bb02f","observation_id":"80e6afbd-2b8b-4f40-bef8-226678459fb1","resolution":{"observed_at":"2026-05-18T10:01:13.465257Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"cited_work":{"arxiv_id":"2504.18208","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.18208","snapshot_observed_at":"2026-06-30T19:55:01.085125Z","title":"Ultra-fast fea- ture learning for the training of two-layer neural net- works in the two-timescale regime.arXiv preprint arXiv:2504.18208","venue":null,"work_id":"74802858-d44a-4c3c-9f04-b14627b4cfb3","year":null},"citing_paper":{"arxiv_id":"2605.15530","last_updated":"2026-05-29T01:50:16Z","snapshot_observed_at":"2026-08-16T19:51:13.144308Z","submitted_at":"2026-05-15T01:59:10Z","title":"Rethinking Neural Network Learning Rates: A Stackelberg Perspective","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-19T14:42:45.648114Z"},"links":{"cited_paper":"/paper/2504.18208","citing_paper":"/paper/2605.15530"},"observation_digest":"sha256:b3c280e4e29416453d57de95283a739f6fd87b6d31aefc22d145e55092cbec90","observation_id":"4d457766-2f47-4506-ad5a-0a628c2b9607","resolution":{"observed_at":"2026-05-19T14:43:06.662422Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"cited_work":{"arxiv_id":"2504.18208","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.18208","snapshot_observed_at":"2026-06-30T19:55:01.085125Z","title":"Ultra-fast fea- ture learning for the training of two-layer neural net- works in the two-timescale regime.arXiv preprint arXiv:2504.18208","venue":null,"work_id":"74802858-d44a-4c3c-9f04-b14627b4cfb3","year":null},"citing_paper":{"arxiv_id":"2605.15530","last_updated":"2026-05-29T01:50:16Z","snapshot_observed_at":"2026-08-16T19:51:13.144308Z","submitted_at":"2026-05-15T01:59:10Z","title":"Rethinking Neural Network Learning Rates: A Stackelberg Perspective","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-30T19:53:45.109784Z"},"links":{"cited_paper":"/paper/2504.18208","citing_paper":"/paper/2605.15530"},"observation_digest":"sha256:7ba885a6830c3b9e54e577ad78a5a7903680aa47e9e4b7c528e7c0e9169f8522","observation_id":"8a9fc212-4fed-4f96-90db-b4893dc3518d","resolution":{"observed_at":"2026-06-30T19:55:01.087709Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.18208","snapshot_observed_at":"2026-07-13T06:19:30.027337Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime, July 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.08843","last_updated":"2026-07-09T18:04:25Z","snapshot_observed_at":"2026-08-13T06:53:26.202598Z","submitted_at":"2026-07-09T18:04:25Z","title":"How are linear representations learned? Exact solutions to the dynamics of abstraction","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-07-13T06:19:30.027337Z"},"links":{"cited_paper":"/paper/2504.18208","citing_paper":"/paper/2607.08843"},"observation_digest":"sha256:78cf7068173ed965895485a3e0724ea5d0904aff4f403625c5acb2aecfe8aced","observation_id":"426ea73a-23d6-4685-9315-7ced04b2844a","resolution":{"observed_at":"2026-07-13T06:19:30.027337Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2504.18208/citation-record","integrity":"/paper/2504.18208/integrity","json":"/paper/2504.18208/citation-record.json","paper":"/paper/2504.18208"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:07.327201Z","title":"A convergence theory for deep learning via over-parameterization","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.327201Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:93d2f1c81185cff27288bc0e0e0a25a81ea85016b7dcba9b0225b942d3b2c284","observation_id":"43c8d84a-ddbb-428b-a381-c984c82433f5","resolution":{"observed_at":"2026-08-16T10:35:07.327201Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:07.331601Z","title":"Gradient flows: in metric spaces and in the space of probability measures","venue":null,"work_id":null,"year":2008},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.331601Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:21c32904eda113cec03c59e1120820f13c2a3f84b7e4575f3242079debbcfb52","observation_id":"af8302f8-7f7f-4c76-9e2e-460ac259b5f0","resolution":{"observed_at":"2026-08-16T10:35:07.331601Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:07.335445Z","title":null,"venue":null,"work_id":null,"year":2000},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.335445Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:27639d92c1fcbea91962660ae4375fb91bc719aa348b00c1849f668d74a53abf","observation_id":"28f3f7b2-7c55-477b-8303-89e4ac102fec","resolution":{"observed_at":"2026-08-16T10:35:07.335445Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:07.339368Z","title":"Maximum mean discrepancy gradient flow","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.339368Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:2543c2fbb98ebb1f10c2777ca7a2a8aca13bf3e8b66ee08e6b3431c0a8089800","observation_id":"f5e42f21-edbd-441f-938a-01d0049d281d","resolution":{"observed_at":"2026-08-16T10:35:07.339368Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:07.343221Z","title":"Breaking the curse of dimensionality with convex neural networks","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.343221Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:b73e888a8d32d6f70cf1da5bc2c4537dd109a807ae72ad79735ae04eb5cdcf7c","observation_id":"b2a9a907-4570-4a8f-b47b-395d1cad14b2","resolution":{"observed_at":"2026-08-16T10:35:07.343221Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.08084","last_updated":"2021-10-15T13:25:32Z","snapshot_observed_at":"2026-08-16T17:49:00.811494Z","submitted_at":"2021-10-15T13:25:32Z","title":"Gradient Descent on Infinitely Wide Neural Networks: Global Convergence and Generalization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.08084","snapshot_observed_at":"2026-08-16T10:35:07.347184Z","title":"Gradient descent on infinitely wide neural networks: Global convergence and generalization","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.347184Z"},"links":{"cited_paper":"/paper/2110.08084","citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:aa666941aa9467dfb1ab0e698d234799d433967ea7f7aa8b64853206ccb6f487","observation_id":"8ba05473-12e8-4ec2-84f1-e29fd8a8382f","resolution":{"observed_at":"2026-08-16T10:35:07.347184Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:07.351646Z","title":"Multiple kernel learning, conic duality, and the SMO algorithm","venue":null,"work_id":null,"year":2004},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.351646Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:83292ad515729c3e51682e6b19bdc4d0c1769dafd96a9201800e9bd994014922","observation_id":"e7c2d240-ba6d-4709-93e4-f189296b3431","resolution":{"observed_at":"2026-08-16T10:35:07.351646Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:07.355445Z","title":"On global convergence of ResNets: From finite to infinite width using linear parameterization","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.355445Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:7035cf8f1471bda2eef8c31cad18db33b0e4282be69a1be48f90eb986327d206","observation_id":"3c49d363-2981-4599-adbc-cceee9a6d60d","resolution":{"observed_at":"2026-08-16T10:35:07.355445Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12887","last_updated":"2025-07-21T14:49:38Z","snapshot_observed_at":"2026-08-16T14:08:19.413848Z","submitted_at":"2024-03-19T16:34:31Z","title":"Understanding the training of infinitely deep and wide ResNets with Conditional Optimal Transport","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12887","snapshot_observed_at":"2026-08-16T10:35:07.359299Z","title":"Understanding the training of infinitely deep and wide resnets with conditional optimal transport","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.359299Z"},"links":{"cited_paper":"/paper/2403.12887","citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:53d2c3f6c0fff667532dcce49da42ecde33b9bf9739e9a78fad376e642ae61eb","observation_id":"d1312402-b2b1-4671-baa6-51b3284792fc","resolution":{"observed_at":"2026-08-16T10:35:07.359299Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:07.363639Z","title":"Modern regularization methods for inverse problems","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.363639Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:9ca7da27d69cd188f379cc98570c4ca29f1ced1045493c232d88f029ab6a535c","observation_id":"aa46ae12-ca97-4ee4-a363-ec46c83010b7","resolution":{"observed_at":"2026-08-16T10:35:07.363639Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:07.367499Z","title":"Learning time-scales in two-layers neural networks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.367499Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:80ca5de090f8c82717f7675e2a8486c083c4d3a7951c281284f2284fd805da43","observation_id":"3cea5307-7035-4f56-8210-ed291b3d441f","resolution":{"observed_at":"2026-08-16T10:35:07.367499Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.19793","last_updated":"2023-11-02T17:33:13Z","snapshot_observed_at":"2026-08-16T14:47:31.525473Z","submitted_at":"2023-10-30T17:55:28Z","title":"On Learning Gaussian Multi-index Models with Gradient Flow","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.19793","snapshot_observed_at":"2026-08-16T10:35:07.371474Z","title":"On learning gaussian multi-index models with gradient flow","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.371474Z"},"links":{"cited_paper":"/paper/2310.19793","citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:55f504f7e780b237b22ae91293fd03d57ab953fcca7a7694eccbf56e04c67efd","observation_id":"7c2c8f5a-eb89-400c-834a-71dd3c2e2532","resolution":{"observed_at":"2026-08-16T10:35:07.371474Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:07.375645Z","title":"Stochastic approximation: a dynamical systems viewpoint","venue":null,"work_id":null,"year":2008},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.375645Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:c09be9e04de177896bf584e8b148850e469e4653741c5c2a133f12b646431ad2","observation_id":"1bc7132a-0975-4ef3-94da-5afc065aba35","resolution":{"observed_at":"2026-08-16T10:35:07.375645Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:07.379486Z","title":"Stochastic approximation with two time scales","venue":null,"work_id":null,"year":1997},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.379486Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:622a851365e41d03960979d05657e4ad6f58308b2a983992a2f3150a2499c2ea","observation_id":"d0400c10-ecaf-44d7-9672-603451ed5597","resolution":{"observed_at":"2026-08-16T10:35:07.379486Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:07.383788Z","title":"Optimization methods for large-scale machine learning","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.383788Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:9ed0dc5459a293560c1f8715c5be9160b5840216b77ca365160676f3e858801a","observation_id":"3e3a16b8-7c41-49e1-aed5-69ceca25e65a","resolution":{"observed_at":"2026-08-16T10:35:07.383788Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:07.388607Z","title":"On the global convergence of Wasserstein gradient flow of the Coulomb discrepancy","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.388607Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:85d329d5a898cfbf657976ca3298794b47193adba46f2fde508e85adf6c73faf","observation_id":"855b6961-4784-4bc4-a3dd-4a8729a7ceee","resolution":{"observed_at":"2026-08-16T10:35:07.388607Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1607.01198","last_updated":"2016-07-05T11:43:44Z","snapshot_observed_at":"2026-08-14T21:49:47.098992Z","submitted_at":"2016-07-05T11:43:44Z","title":"Quantization of Measures and Gradient Flows: a Perturbative Approach in the 2-Dimensional Case","version":1},"cited_work":{"arxiv_id":"1607.01198","doi":null,"metadata_source":"pith","pith_arxiv_id":"1607.01198","snapshot_observed_at":"2026-08-16T10:35:07.801635Z","title":"Quantization of Measures and Gradient Flows: a Perturbative Approach in the 2-Dimensional Case","venue":"math.AP","work_id":"e36edf7e-f4b6-4955-bd4d-e3fe560e2a42","year":2016},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.392370Z"},"links":{"cited_paper":"/paper/1607.01198","citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:acabbf7dd1c464ed671494d4d704daff4bff0a407f01b1e53cc778c34d3543d8","observation_id":"5e206f2e-96a7-4c32-8454-e6ace54bbecf","resolution":{"observed_at":"2026-08-16T10:35:07.808148Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:07.396185Z","title":"(De)-regularized Maximum Mean Discrepancy Gradient Flow","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.396185Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:134e5e46fda16c2e31e61e758f35d4b01918ce937bef98b0857c3ae694af783a","observation_id":"dff33a19-076b-4504-9c7d-626bceb54753","resolution":{"observed_at":"2026-08-16T10:35:07.396185Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:07.399853Z","title":"Analysis of langevin monte carlo from poincare to log-sobolev","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.399853Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:0da63e745f1fce275fa13540ace23dfd38d5bc00b5af73c7c2ab7e4e284770a7","observation_id":"e1ed5625-643a-45a5-880b-c52e85306d2f","resolution":{"observed_at":"2026-08-16T10:35:07.399853Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:07.403617Z","title":"SVGD as a kernelized Wasserstein gradient flow of the chi-squared diver- gence","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.403617Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:94f3d299213eed3f75d884798cb02a3c9d1e08309b7d8c6c51826ff50c6e7fce","observation_id":"67cfadc8-c3b6-4569-8d8a-df5313d741a8","resolution":{"observed_at":"2026-08-16T10:35:07.403617Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.777039Z","title":"Mean-Field Langevin Dynamics: Exponential Convergence and Annealing","venue":null,"work_id":"364d65a8-b5a3-4cd4-9d67-b3a79b76ed1b","year":2022},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.407261Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:2a682ed1918b9303eaee4be4829e36a0fbc0e819e7733d1b7d1bad326bdd2cc2","observation_id":"7a7398c6-0fc6-4a7f-8f74-1c3f5994871f","resolution":{"observed_at":"2026-08-16T10:35:08.781067Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.765068Z","title":"On Lazy Training in Differentiable Pro- gramming","venue":null,"work_id":"4e1e65a0-3381-409a-96d2-e9cc9a47fbd7","year":2019},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.411104Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:99443e3bbeaeda47e04548de681cdb03d7398f7e5dda3c660377ccca74ca5fd4","observation_id":"ff209de8-3bcc-464c-9462-4be7fa626d91","resolution":{"observed_at":"2026-08-16T10:35:08.768816Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.752849Z","title":"On the Global Convergence of Gradient Descent for Over- parameterized Models using Optimal Transport","venue":null,"work_id":"f1e9d694-0999-477a-bc92-86d221486445","year":2018},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.414632Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:f5ff79cc8e4a08af454830a3d0ce2dd4faf7150dcbea1594f46a6c69ab4cba7a","observation_id":"6cad4cfd-c76c-4b77-8a29-2f94636d71d3","resolution":{"observed_at":"2026-08-16T10:35:08.756748Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:07.418212Z","title":"Approximation by superpositions of a sigmoidal function","venue":null,"work_id":null,"year":1989},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.418212Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:cc2299253e8e6024e15d9676b4d81d43dea22605100a29c11717d51ae7cf4181","observation_id":"8333ab3d-e1c4-4f60-8aab-64f81c22d1cb","resolution":{"observed_at":"2026-08-16T10:35:07.418212Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.734516Z","title":"Exact reconstruction using Beurling minimal ex- trapolation","venue":null,"work_id":"a080c440-70df-494a-a96d-33a0668001f2","year":2012},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.421866Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:8765bc659d552c8f3d52827f1d4076ac790ceecb3aba92941fce953054c0b64d","observation_id":"db6a92df-79e1-4424-b247-f3d7a55d7b83","resolution":{"observed_at":"2026-08-16T10:35:08.738200Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.723813Z","title":"High-dimensional data analysis: The curses and blessings of dimen- sionality","venue":null,"work_id":"c0ba7a91-d47d-443b-8f56-e0aa2aa5c893","year":2000},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.425492Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:a1cbc8e2d2381b25e6563ab594689227872064dfdb8d0700df10548d313ccb21","observation_id":"841dc573-15d0-4049-93b1-422fe9174208","resolution":{"observed_at":"2026-08-16T10:35:08.727753Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.712732Z","title":"Gradient descent finds global minima of deep neural networks","venue":null,"work_id":"71697ed9-a938-46dc-9aba-b4cebdd0e0bc","year":2019},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.429006Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:90e365e4ec81b444f58440241209b394937cd8b723d385739cd0d23eed886aff","observation_id":"fd6f6990-ef30-4bd0-adc7-f6fa1df90205","resolution":{"observed_at":"2026-08-16T10:35:08.716819Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.701291Z","title":"Exact support recovery for sparse spikes deconvolution","venue":null,"work_id":"b2df4238-3081-4843-bfb0-59118973e9f9","year":2015},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.432865Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:9973ed421cb90ff58689c63dabb23d61f59d1117d25b80899d5c995e876da196","observation_id":"70e46147-6316-4327-b4cf-d280160e4282","resolution":{"observed_at":"2026-08-16T10:35:08.705379Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.689673Z","title":"On the rate of convergence in Wasserstein distance of the empirical measure","venue":null,"work_id":"9d468657-768c-48b6-a9cd-e315eb5bace7","year":2015},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.436437Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:f22ba12e0382b72779dea4e0acb42358bb966075d6b456a19bad08f9f6e9a287","observation_id":"52c82066-5453-4c7f-9fe7-7f84da4c1c45","resolution":{"observed_at":"2026-08-16T10:35:08.693924Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.677501Z","title":"Global convergence in training large-scale transformers","venue":null,"work_id":"db2ce5be-eeee-4bca-8866-90981246fc91","year":2024},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.439736Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:b60e4c4990f8a8b52a9c2f6dfaa05a8aa06db19cd2a6ef2ef1af8c024fe2a330","observation_id":"d94f6a5c-b8ac-45b3-aca2-eede52f97123","resolution":{"observed_at":"2026-08-16T10:35:08.681588Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.666113Z","title":"When do neural networks outperform kernel methods?","venue":null,"work_id":"ca31cd1e-3823-4444-a2e9-ad453ff8a6e6","year":2020},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.443333Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:9b221f0fcc362960f5403cca7431d75d5e2015014a04a8eedaaa653dcd81f82c","observation_id":"70195d83-b7ea-42c4-bbbb-f76069189c68","resolution":{"observed_at":"2026-08-16T10:35:08.670087Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.653390Z","title":"KALE flow: A relaxed KL gradient flow for probabilities with disjoint support","venue":null,"work_id":"69f8b918-12d0-40c9-a30f-2027289cb754","year":2021},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.447113Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:45d4427f49098f583067864d700a7aa3116edde1634e7387d397fc0fd7fbcf3a","observation_id":"bd5aa5cf-2e7d-473a-99ae-138d353ea47b","resolution":{"observed_at":"2026-08-16T10:35:08.658045Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.640499Z","title":"Separable nonlinear least squares: the variable projection method and its applications","venue":null,"work_id":"02f9cd8e-1f0f-4a86-9dac-19c1b8408c27","year":2003},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.450874Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:5710304e64b0b1c2fa83ebf665a501904df994c7c733cc2e7e151e83edf120dc","observation_id":"bbf6f600-518f-432a-83e6-37a870676ed6","resolution":{"observed_at":"2026-08-16T10:35:08.644291Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.628848Z","title":"The differentiation of pseudo-inverses and nonlinear least squares problems whose variables separate","venue":null,"work_id":"891f2b69-4417-4c5d-9a05-437e60806221","year":1973},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.454843Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:432c0a6b103eb61bd614f89bc0c6ce4c6c10907121c66d8dbf7db8e7133602dc","observation_id":"95a47204-1407-471b-9341-46657ed91a66","resolution":{"observed_at":"2026-08-16T10:35:08.632692Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.617404Z","title":"Deep Learning","venue":null,"work_id":"1e7dc8e4-fc7b-44f6-965b-0cc485a23014","year":2016},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.458726Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:09406381b6cad66292ac2e7b3b810281a883c9ce8c0122c5970195e31ada353a","observation_id":"dd835aff-7ccc-4e58-b891-61340e6fd779","resolution":{"observed_at":"2026-08-16T10:35:08.621414Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.605588Z","title":"A kernel two-sample test","venue":null,"work_id":"7bf783ee-f831-40f6-ae24-52060f8efdd0","year":2012},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.463363Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:2ae6fcb1800b08b9ee58a7331faaaec4b3a3f2e93a86f2a2010fe0d75b95309b","observation_id":"c791fc03-04ba-4da0-ac7f-0a5184e42e5b","resolution":{"observed_at":"2026-08-16T10:35:08.609644Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.592674Z","title":"Shampoo: Preconditioned stochastic tensor optimization","venue":null,"work_id":"110bd53f-9858-4bed-83ee-5ac7db335bae","year":2018},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.467918Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:ad3936b6f2313b013de142fd11ece53d0ea9df0986832d91ed033312de9a25ba","observation_id":"56477777-4d41-4520-839d-5607d8bd3dcc","resolution":{"observed_at":"2026-08-16T10:35:08.597707Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:07.473822Z","title":"Ordinary differential equations","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.473822Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:280213327ce76fc79d6329610799510ba209644d342d58460bdd1ce2f85f6e1b","observation_id":"03311575-1a82-4ed7-b77f-8ac6cedf3ded","resolution":{"observed_at":"2026-08-16T10:35:07.473822Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.571674Z","title":"Deep residual learning for image recognition","venue":null,"work_id":"fa847e73-e0e9-4c4f-8875-7c10cb6a9069","year":2016},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.478925Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:037f4c3e70b2eb45204080db82c89188fdfbc91eb2ac8a6f5c64d4cec900513f","observation_id":"130ba544-5796-4a29-b037-49485d8e0cad","resolution":{"observed_at":"2026-08-16T10:35:08.575716Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.559926Z","title":"Generative Sliced MMD Flows with Riesz Kernels","venue":null,"work_id":"ef5bf06d-567f-4832-bbb7-a14cbdbbec80","year":null},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.483586Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:9ebe7efde9bb8f32bc408d35a4d800cc57929fa570b4a2db5152b3287d43352b","observation_id":"d7cfa46c-5914-4c2d-ac26-aaca7e475a08","resolution":{"observed_at":"2026-08-16T10:35:08.564294Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.546784Z","title":"Wasserstein gradient flows of the discrepancy with distance kernel on the line","venue":null,"work_id":"a848aca9-9b94-494a-b222-1fac070f22e1","year":2023},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.487878Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:bd91497e7f8f25eb2f1fc6b20e8e0691ed9439a3d7be35bcdb94d874e1c81ec1","observation_id":"1fdf15e9-c590-4f91-bf5b-6691a359d317","resolution":{"observed_at":"2026-08-16T10:35:08.551132Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.534614Z","title":"Wasserstein steepest descent flows of discrepancies with Riesz ker- nels","venue":null,"work_id":"04ba7f49-916d-4513-98e5-06850f5fef13","year":2024},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.491761Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:dd2cfa9fa959ad03ad8712c0103f96bcd45313d5caf904dce7e094b5da3f913e","observation_id":"1a27aad9-9fcc-4bf5-b6cb-19973e7e515c","resolution":{"observed_at":"2026-08-16T10:35:08.538519Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.521894Z","title":"ODEPACK, a systemized collection of ODE solvers","venue":null,"work_id":"982ba588-c4f8-48bf-a65f-bae0f47bc784","year":1983},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.495753Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:287c1e1901068eeaf6cf71c7d0194b5f3f3091284d99db352a2850e1d3ce0919","observation_id":"113e99b3-790c-40ee-bb77-25d140cded28","resolution":{"observed_at":"2026-08-16T10:35:08.526539Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.508901Z","title":"Kernel methods in machine learning","venue":null,"work_id":"a406b52d-a41a-4a5e-846b-ab10376f3b20","year":2008},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.499718Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:44736f173cf770dfd22f3e6c7d756cc32532a90c24d1b73fea348195ad43feba","observation_id":"d122c635-f9f3-4494-afa5-eb33601b47db","resolution":{"observed_at":"2026-08-16T10:35:08.513034Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.494782Z","title":"Mean-field Langevin dynamics and energy landscape of neural networks","venue":null,"work_id":"65d91e2e-b6e1-4fe4-806d-8b9d38031f6f","year":2021},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.503539Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:ca7f5e6e7070815ac2178db3e6d2d3799e1add69d8345b21d06c58ae432e8c5a","observation_id":"83363099-aa87-40b3-a46e-a819b317fd5a","resolution":{"observed_at":"2026-08-16T10:35:08.499043Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.481027Z","title":"Asymptotic analysis for a very fast diffusion equation arising from the 1D quantization problem","venue":null,"work_id":"b54e0c6b-0c11-4629-951c-a9389da50353","year":2019},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.507970Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:dcf03b265dfa6f908e84595c33a15635795eff1aec0ac674920cd0b848b7d1c9","observation_id":"968b916d-1e30-4574-85e1-2bdd5318e50f","resolution":{"observed_at":"2026-08-16T10:35:08.485512Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.469606Z","title":"Weighted ultrafast diffusion equations: from well-posedness to long-time behaviour","venue":null,"work_id":"744499e5-feb9-4b8d-8c49-dc3eb9586258","year":2019},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.512328Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:11a945bb693b4d27b51aac52111f8e0118186df699971b07f2f1d48c559626c1","observation_id":"3944f926-b318-4e72-be94-589c56a512f5","resolution":{"observed_at":"2026-08-16T10:35:08.473408Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.457519Z","title":"A note on convergence of solu- tions of total variation regularized linear inverse problems","venue":null,"work_id":"66e6dbcd-c43f-4144-8935-9ac1bf9785c3","year":2018},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.518621Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:3304b4970f01f41024ee5f847295de866448f50bffd2f7a53858860d776a4d46","observation_id":"f702c31b-0e03-4da6-afe0-ff10bcdead05","resolution":{"observed_at":"2026-08-16T10:35:08.461701Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.445942Z","title":"Neural tangent kernel: Convergence and generalization in neural networks","venue":null,"work_id":"a73566aa-ac03-4859-9e61-5b6ce387f998","year":2018},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.523324Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:6a1e03e2eabe644c691a9a2dad26d9dbeda8e3cde59bf45c677ef0bcac3b799c","observation_id":"ed7981e3-2b6e-4a60-81ec-26e942282705","resolution":{"observed_at":"2026-08-16T10:35:08.450155Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.432893Z","title":"The variational formulation of the Fokker–Planck equation","venue":null,"work_id":"d41d554a-e788-4887-b8b3-3d15b7d48510","year":1998},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.527421Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:d5d7b432db4eac8e30634c3bf0f973c09a68b03e45e737e53415e242e2dafb40","observation_id":"02e28904-395b-4f75-8938-1eedb6c7f931","resolution":{"observed_at":"2026-08-16T10:35:08.437018Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.421069Z","title":"Radial basis function neural network training using variable projection and fuzzy means","venue":null,"work_id":"d327b8ce-e117-4f2f-b339-6d4c71e68bd2","year":2024},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.531171Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:4d1d929a95b4df15fa4fcda50339ceec76ee41d67c0aed0b40aa769ff5688e44","observation_id":"60dbeb9e-e387-4da6-b687-223afdb3f42b","resolution":{"observed_at":"2026-08-16T10:35:08.425070Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.409287Z","title":"Learning multiple layers of features from tiny im- ages","venue":null,"work_id":"eab7e913-ea55-45f5-b957-f9d6f3504331","year":2009},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.535597Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:fce3c5068e527c9e5c7d92b9844a1cee7a7b1fae566a8ffd1f51ce1df518d327","observation_id":"3ccc1c67-2b08-487a-a756-45a8192ab7cf","resolution":{"observed_at":"2026-08-16T10:35:08.413514Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.396942Z","title":"Learning the kernel matrix with semidefinite programming","venue":null,"work_id":"d936f0e1-1e78-491b-b6b0-4386849201eb","year":2004},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.539194Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:94e477e6de2c5a563f27679f187c137d2a7ca502adcfe6837eb0f6f2c3bb3549","observation_id":"2e1c0806-579f-46f1-a586-05217f4fc044","resolution":{"observed_at":"2026-08-16T10:35:08.401583Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.381686Z","title":"Wide neural networks of any depth evolve as linear models under gradient descent","venue":null,"work_id":"b0f3c5c5-4aac-4ca8-8b98-a1306a624d06","year":2019},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.542782Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:96e1adb63b713c6a1c3ff373cc244f063bfb2f60b7a5b1a67b8860c17949bfcc","observation_id":"5028a977-f57a-4f1e-8c14-542364e79a5c","resolution":{"observed_at":"2026-08-16T10:35:08.386361Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.370264Z","title":"Optimal entropy-transport prob- lems and a new Hellinger–Kantorovich distance between positive measures","venue":null,"work_id":"fb1f1d53-a336-4a59-b639-b66b18a16e06","year":2018},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.546230Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:9daaa48ea1e76be9b5d7d77ee7ffe48d09029a7bbffbfef3a3ceea9295734e74","observation_id":"76339589-fd8f-49f6-8915-7a21b9633d30","resolution":{"observed_at":"2026-08-16T10:35:08.374069Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.355360Z","title":"On the linearity of large non-linear models: when and why the tangent kernel is constant","venue":null,"work_id":"886ecd04-bc8d-4677-a180-e2a1f1f00ef6","year":2020},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.549915Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:c99eb55b10d4207cd7c0ba3536eaeb322c72f069d8e914160e98ce4f87e39c80","observation_id":"2fb08b93-6058-4b87-8925-853dae03d768","resolution":{"observed_at":"2026-08-16T10:35:08.361643Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.342812Z","title":"Leveraging the two-timescale regime to demonstrate convergence of neural networks","venue":null,"work_id":"b5244cde-c240-437e-a1df-70f44a243ef7","year":2023},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.553389Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:d09ae3f1852513fbdc2c32c5652eb74841620609e9b8c3d255710fa8083cf916","observation_id":"14541fe5-765b-43c0-9d2b-f1829e01a933","resolution":{"observed_at":"2026-08-16T10:35:08.346724Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.331193Z","title":"Mean-field theory of two-layers neural networks: dimension-free bounds and kernel limit","venue":null,"work_id":"439bf6cb-3c1a-4892-8bed-ae9db6eea283","year":2019},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.557746Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:ef537793533da1ba96a7654614ba0fc98fae2eafeb21feeb6b74cf25dd0ca74d","observation_id":"173241b4-afcb-4c06-8ce5-114c48d4c3d0","resolution":{"observed_at":"2026-08-16T10:35:08.335113Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.318759Z","title":"Universal Kernels","venue":null,"work_id":"d8dc8e4f-ba3e-4a0b-9bce-65fe3e463bc9","year":2006},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.561253Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:086e54c79327ffa20e10bbfec1677f56d4c418345aee74412131ac40cd8ec9c8","observation_id":"e62cbbc6-84a0-4d32-837a-3b90b7ae8ae4","resolution":{"observed_at":"2026-08-16T10:35:08.323034Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.305066Z","title":"Envelope theorems for arbitrary choice sets","venue":null,"work_id":"a16c7c49-15b9-4268-8c8d-33538b0630d9","year":2002},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.565111Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:060876192d9910040bcb1d78305136538161db2db124ef660b80094d47659128","observation_id":"b9d797a1-fb9c-4474-a556-6d9af415e119","resolution":{"observed_at":"2026-08-16T10:35:08.309707Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.290297Z","title":"Kernel mean embedding of distributions: A review and beyond","venue":null,"work_id":"2830c360-644f-41ad-a446-feec66b87705","year":2017},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.568505Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:907fba7f8f0de1fdae94c73783050efd1894a566580f96a6fb848f22ba8b1654","observation_id":"3907dbe5-9522-4eba-9ae3-c1b81039796c","resolution":{"observed_at":"2026-08-16T10:35:08.294975Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04613","last_updated":"2025-04-11T12:00:09Z","snapshot_observed_at":"2026-08-16T14:20:44.256137Z","submitted_at":"2024-02-07T06:30:39Z","title":"Wasserstein Gradient Flows for Moreau Envelopes of f-Divergences in Reproducing Kernel Hilbert Spaces","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.04613","snapshot_observed_at":"2026-08-16T10:35:07.572108Z","title":"Wasserstein Gradient Flows for Moreau Envelopes of f-Divergences in Reproducing Kernel Hilbert Spaces","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.572108Z"},"links":{"cited_paper":"/paper/2402.04613","citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:bbf87aecf326d520f1ca92b2e9f7714029a33178415fdeed8254b541639d078a","observation_id":"3aeaea96-7700-456d-9479-511aaadfa0c7","resolution":{"observed_at":"2026-08-16T10:35:07.572108Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.274761Z","title":"Train like a (Var) Pro: Efficient training of neural networks with variable projection","venue":null,"work_id":"b23835d1-bb89-4f52-a93b-f7e3631b7afd","year":2021},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.576097Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:e161256e05e6ac385e7f99fba6b94cb9ca3adcfd3efd161402551d4c75a5bde2","observation_id":"d578c376-bc6f-41bc-9478-2d46b0ef6cd7","resolution":{"observed_at":"2026-08-16T10:35:08.279066Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.262202Z","title":"Convex analysis of the mean field langevin dynamics","venue":null,"work_id":"97d3d3bf-16cd-4534-915b-28a9d779c5c9","year":2022},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.579844Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:65a1da2ff86945aa1af2fdf8b950e41d90d0f7bfb0785355ca40b0c60de52842","observation_id":"6d676dde-e413-4470-b262-2d5808102cc6","resolution":{"observed_at":"2026-08-16T10:35:08.266273Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.249972Z","title":"Separable least squares, variable projection, and the Gauss-Newton algorithm","venue":null,"work_id":"d34e15f8-9bad-4649-925d-c48da7e92ace","year":2007},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.583566Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:78157ce72d3b52759cf74dcd1bf3db655745e350a67a93a64ae2c4bb3b64fdfe","observation_id":"a900796e-1e0c-4916-9e5e-1e93f1e659f6","resolution":{"observed_at":"2026-08-16T10:35:08.253972Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.239441Z","title":"Stochastic processes and applications","venue":null,"work_id":"9b4e2359-edc5-4b45-b628-9e3370be22b3","year":2014},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.587258Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:33f31ca869ad9825e572959624f86c7dc6041bf430d9e963a3330d6ab4369daa","observation_id":"c43e0866-4b56-439d-81c1-db6bb0fc9a91","resolution":{"observed_at":"2026-08-16T10:35:08.242956Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.228122Z","title":"An optimal Poincar´ e inequality for convex do- mains","venue":null,"work_id":"457695a4-2b37-44c2-9356-ebf7032d6edf","year":1960},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.590672Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:86c8597b5433a9ec1b826b24dc78126e475f6e0c460873bd94dfb7c422f722f1","observation_id":"c0487041-6fea-45c7-bf52-91ec8ebbc026","resolution":{"observed_at":"2026-08-16T10:35:08.232142Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.216830Z","title":"Variable projections neural network training","venue":null,"work_id":"988bc5c7-95d5-41ec-97f4-0f00c33169e5","year":2006},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.594401Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:3534e9544ed9b067cb7691ff0cafa2f3496d12ea8df155df6007d8206668db99","observation_id":"c0187158-cd7b-431a-9dc1-f3cc07d356c1","resolution":{"observed_at":"2026-08-16T10:35:08.221072Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.204039Z","title":"Duality and stability in extremum problems involving convex functions","venue":null,"work_id":"c14d2765-ad3d-43a5-9314-7477243ada61","year":1967},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.597825Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:5ccec66ecdd8aa69cb65f663acbd06c19538bceb8fe4df0a4be07d7db8ecc7d5","observation_id":"a14b27a1-3acd-4f7c-b6d7-dd4196ed960e","resolution":{"observed_at":"2026-08-16T10:35:08.208399Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.192404Z","title":"Integrals which are convex functionals","venue":null,"work_id":"724271e2-27f2-4e4d-96a3-08606ab34402","year":1968},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.601175Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:8b8eecb7585c8b94f5954b04122bf8b4db70941c036601ee8c3b13bd04edaa92","observation_id":"475649ef-f39a-4837-ae6a-9fb5425651db","resolution":{"observed_at":"2026-08-16T10:35:08.196178Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.181105Z","title":"Integrals which are convex functionals. II","venue":null,"work_id":"4296aa95-fa4e-4004-85a8-f0f82c1de54d","year":1971},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.604658Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:97437ad093068dc8f0fe3af01a5902c72ef86495e38a31e53bd904f091291473","observation_id":"edd5cc63-e611-453d-b8e9-d7a43ba25e34","resolution":{"observed_at":"2026-08-16T10:35:08.184693Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.170112Z","title":"Global convergence of neuron birth-death dynamics","venue":null,"work_id":"d155ab0a-92c1-4d62-9bba-b4fc06cdaf0d","year":2019},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.608371Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:c8a514fc939b275f05c191e3aa9863bca72a63b1f97bfb1679d7140ec913b493","observation_id":"a276b96f-a913-4fc7-b9ec-22d7ceb89d99","resolution":{"observed_at":"2026-08-16T10:35:08.173918Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.158568Z","title":"A Course in the Calculus of Variations: Optimization, Regularity, and Modeling","venue":null,"work_id":"7036986d-4237-4acb-8d44-e7e1bfb36158","year":2023},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.611968Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:491b85c9a1aee98f22eb339420ef0f4e1ef705d9097cdba491838d7fa7a5555f","observation_id":"cd28db76-5b91-45dd-9f61-a1fd9f740fc0","resolution":{"observed_at":"2026-08-16T10:35:08.162643Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.147620Z","title":"{Euclidean, metric, and Wasserstein} gradient flows: an overview","venue":null,"work_id":"f2cb7e4d-3c09-4eeb-8885-61b298f7462a","year":2017},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.615478Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:5172854035b10da6c206f8f9df0c3a233e5e61dd815dd055642394f0ee92bc60","observation_id":"155d3453-412f-4783-9caf-3f4b6cb585fc","resolution":{"observed_at":"2026-08-16T10:35:08.151424Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.137130Z","title":"Optimal transport for applied mathematicians","venue":null,"work_id":"01ec6b1d-feec-43b3-9851-feff7d4b9fde","year":2015},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.618972Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:a27704073e1ce2c52b1b7cc2da49996643b6125a2f3e3f47615bbbfde21e954e","observation_id":"45f5b3d0-0937-4d6f-b3e8-761a9576fb28","resolution":{"observed_at":"2026-08-16T10:35:08.140736Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.125732Z","title":"Learning with kernels: support vector machines, regularization, optimization, and beyond","venue":null,"work_id":"16572f1b-2ea7-4e59-832f-1fc4fa09d5a4","year":2002},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.622897Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:073499cdd9f3dbf6746db3f3352e362aeea681cdd2542f5fc1c1b922c6f0f60b","observation_id":"23f044b0-a27f-4d33-9859-b9dc218d2721","resolution":{"observed_at":"2026-08-16T10:35:08.129853Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.114605Z","title":"Equivalence of distance-based and RKHS-based statistics in hypothesis testing","venue":null,"work_id":"ec09a070-c097-4454-9267-3b962eef5c67","year":2013},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.626206Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:7c1003a3a3039cb54bce9a98e84c7898edbc3d134a6147ac8f460a2b43d7c1e1","observation_id":"bd6521b1-dd81-4875-b098-67cbed6f7a8f","resolution":{"observed_at":"2026-08-16T10:35:08.118507Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.103463Z","title":"Mean field analysis of neural networks: A central limit theorem","venue":null,"work_id":"9a3b201c-7939-4f45-9f3c-c27e62e54f52","year":2020},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.629731Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:51ba848a631239a32ee61f9f900c21d869fb4199be0b730c9b33dd9bec338d57","observation_id":"d3e879b2-42e5-42a4-8bed-c5850c625000","resolution":{"observed_at":"2026-08-16T10:35:08.107461Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.092126Z","title":"Separable non-linear least-squares minimization-possible improvements for neural net fitting","venue":null,"work_id":"569921a6-bb7d-4721-8e83-b312569ce6b2","year":1997},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.633277Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:634f57dde63816f2300bc160fc711021a62f83034e05fb3a4818d5627ea32e96","observation_id":"172fa144-6266-4a13-9432-d479361960ad","resolution":{"observed_at":"2026-08-16T10:35:08.095962Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.081417Z","title":"Universality, Char- acteristic Kernels and RKHS Embedding of Measures","venue":null,"work_id":"d905b55e-9d08-4090-aef0-040e4790dd51","year":2011},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.636710Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:6a4364d905ef5378956bf61bf70af22a33182ad904a2335f3b30cd68d0303431","observation_id":"c6ad61f1-97eb-4d86-92b6-dc2851b9aef7","resolution":{"observed_at":"2026-08-16T10:35:08.085247Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.070229Z","title":"Support vector machines","venue":null,"work_id":"516a0621-d07c-493d-8498-453799f280cc","year":2008},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.640241Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:a89f59acbdb8aae0894b28ecdf9a2e4c52870a02f6e2ed57fa7d044d2c869ccc","observation_id":"0c8ea6bc-f248-402a-811c-5940b2da0737","resolution":{"observed_at":"2026-08-16T10:35:08.073897Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.057331Z","title":"Mercer’s theorem on general domains: On the interaction between measures, kernels, and RKHSs","venue":null,"work_id":"18bee261-bcad-4d64-96b2-9879d47fe527","year":2012},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.644028Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:05d3f6c9eed9cd7805a060307027923605234a63e3f108b382351150d9b6c98f","observation_id":"1c60b681-6673-4a64-85b8-dcd257919351","resolution":{"observed_at":"2026-08-16T10:35:08.061846Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.045651Z","title":"Random Features Methods in Supervised Learning","venue":null,"work_id":"4b5b533d-df0f-46ba-b742-f01f2364551c","year":2019},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.647790Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:18d645cb1add3bf5fb7d0fd448df7a875b0ba97c7afca1b90c550174fb347875","observation_id":"2ba8fc1d-3810-4825-9b24-9bdb1a10ad22","resolution":{"observed_at":"2026-08-16T10:35:08.049529Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.033827Z","title":"Feature learning via mean-field langevin dynamics: classifying sparse parities and beyond","venue":null,"work_id":"53cd6e35-8ecc-4617-bdf2-72c1b0c1691b","year":2023},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.651550Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:0c7006b950b2fc40ac5e74aef71e39f2693a1c95af4122a1b9fec0624d16ec97","observation_id":"3b4bbfaa-265c-4b59-b8d3-4dcb579eab48","resolution":{"observed_at":"2026-08-16T10:35:08.038214Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.022079Z","title":"Mean-field Analysis on Two-layer Neural Networks from a Kernel Perspective","venue":null,"work_id":"e7ed6910-9412-4488-b001-90b9821ea65a","year":null},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.655341Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:8090de32b866df188ed1a4f6ee75d560661eca2f38610223498fa7df3f0ef891","observation_id":"55f53614-caf1-4b57-a219-c281c8635d65","resolution":{"observed_at":"2026-08-16T10:35:08.026763Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:08.010157Z","title":"Smoothing and decay estimates for nonlinear diffusion equations: equa- tions of porous medium type","venue":null,"work_id":"ed0131d3-0efb-4247-b3fa-a9b0863d61b6","year":2006},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.658879Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:005b40f4393fba35219bd4c8661c129ba69ee19ba60ef55c04369563df136b51","observation_id":"b95774df-4dd7-4a94-a5ee-6b569b18829a","resolution":{"observed_at":"2026-08-16T10:35:08.014235Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:07.997954Z","title":"The porous medium equation: mathematical theory","venue":null,"work_id":"d36f3388-3f47-4c16-b72c-ee0d24c89eb9","year":2007},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.662664Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:76f812b28f726ae8a5a98cd3493ef2eb7b2709a42018f2332742659e652ca5ba","observation_id":"dbebf28c-b926-452a-ba28-87a22883e409","resolution":{"observed_at":"2026-08-16T10:35:08.001980Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:07.986320Z","title":"Partial optimization and Schur complement","venue":null,"work_id":"a604c60c-1068-43e7-aede-1aad0bc7273b","year":2019},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.666227Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:0a0df6e8a54e988b9d7add43e41fcf820e5dc44f6c986bdbf1872b75b53fdc00","observation_id":"22a5ca96-f45b-4efc-b7b0-b1d3355f863b","resolution":{"observed_at":"2026-08-16T10:35:07.990677Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:07.974319Z","title":"Optimal transport: old and new","venue":null,"work_id":"d5960ca5-1822-4541-ada9-b99e70b236ed","year":2009},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.670041Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:1f0b86a1da2e494f4d71638054a37ea018eb68dbb08af4dc903e0197f862b122","observation_id":"e14bef91-d744-478f-a048-e409767aba58","resolution":{"observed_at":"2026-08-16T10:35:07.978375Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:07.960342Z","title":"Mean-field langevin dynam- ics for signed measures via a bilevel approach","venue":null,"work_id":"b22accef-9b26-4825-b095-7449c987afa4","year":2024},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.673587Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:c824b93aa6305771122f6bdd8cc408b6ee7c9be539ebb45f730a44b2c71c9637","observation_id":"b03a16f9-c787-4847-a408-a89e6f01ef7a","resolution":{"observed_at":"2026-08-16T10:35:07.964346Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:07.948714Z","title":"Tensor programs iv: Feature learning in infinite-width neural networks","venue":null,"work_id":"379372ba-744a-4240-a7d7-294685003c69","year":2021},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.677238Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:4a8fe68f71a98c07a6241dea843f059a37f84b0774b701c1f301c5b583c65827","observation_id":"49cd9bec-aa2f-4c8e-8491-a238719f9482","resolution":{"observed_at":"2026-08-16T10:35:07.952443Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:07.936000Z","title":"Gradient descent optimizes over-parameterized deep ReLU networks","venue":null,"work_id":"97c71504-9ac1-4a81-9a85-3a3660e5a30f","year":2020},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.681206Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:b089504d9d74b106e07af885941fb19ae108d5f14178d8bb3a482fb214710429","observation_id":"a1435c61-be10-4513-8238-b43adac6039f","resolution":{"observed_at":"2026-08-16T10:35:07.940225Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:07.924144Z","title":"biased” quadratic regularization fb :t7→ 1 2t2 or the “unbiased","venue":null,"work_id":"623f6106-e37f-41c0-a9f8-349a84f0bc80","year":null},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.685279Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:9b65bbb4466036a883110c8c6b5480456d536532d61e0947bc4a105752d4b7e9","observation_id":"11e3a01f-38f5-4156-9660-87aebf4a46b1","resolution":{"observed_at":"2026-08-16T10:35:07.928225Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:35:07.912553Z","title":"(50)) of width M∈{ 32, 128, 512, 1024}","venue":null,"work_id":"c58ed15a-1066-495c-99ad-6e606af3d13f","year":null},"citing_paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime","version":2},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-16T10:35:07.689364Z"},"links":{"citing_paper":"/paper/2504.18208"},"observation_digest":"sha256:a4ed22c6dc99c64f6c2abf4bb2ed6c6217ed948d01484f08dc5743da11ded741","observation_id":"8919791a-ee87-4a5e-b69f-b7a2273071c3","resolution":{"observed_at":"2026-08-16T10:35:07.916189Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2504.18208","last_updated":"2025-07-21T14:02:53Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-16T10:19:40.520110Z","submitted_at":"2025-04-25T09:40:10Z","title":"Ultra-fast feature learning for the training of two-layer neural networks in the two-timescale regime"},"reference_resolution":{"displayed":94,"state_counts":{"malformed_identifier":2,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":22,"verified_exact":1,"verified_fuzzy":69},"total_outbound_references":94},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"thesis":"As of 16 August 2026, this Paper Citation Record lists 94 of 94 outbound references and 5 inbound Pith citation observations for arXiv:2504.18208."}