{"as_of":"2026-08-15T08:28:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3ee761e3cbeef007a5764ba86ef581f4f100169df6b8fc8f8f36ed91e9910e2f","coverage":[{"denominator":79,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":79,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T12:48:46.143996Z","state":"measured"},{"denominator":79,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":79,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2505.23489/citation-record","integrity":"/paper/2505.23489/integrity","json":"/paper/2505.23489/citation-record.json","paper":"/paper/2505.23489"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"1807.04162","last_updated":"2018-10-04T20:26:11Z","snapshot_observed_at":"2026-08-14T18:53:11.453770Z","submitted_at":"2018-07-11T14:39:17Z","title":"TherML: Thermodynamics of Machine Learning","version":3},"cited_work":{"arxiv_id":"1807.04162","doi":null,"metadata_source":"pith","pith_arxiv_id":"1807.04162","snapshot_observed_at":"2026-08-07T12:48:49.850205Z","title":"TherML: Thermodynamics of Machine Learning","venue":"cs.LG","work_id":"38eaf705-202a-4504-87c9-dc68ba2ef35e","year":2018},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:36.572374Z"},"links":{"cited_paper":"/paper/1807.04162","citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:79ad382613aa84bc67bc92a329001d8c17b4591ca13d10dfe7c45614b6011af8","observation_id":"2c5405bf-6824-4880-a578-313cde9c8185","resolution":{"observed_at":"2026-08-07T12:48:50.002829Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.05337","last_updated":"2023-06-07T09:50:21Z","snapshot_observed_at":"2026-08-13T14:08:28.827124Z","submitted_at":"2022-10-11T11:00:04Z","title":"SGD with Large Step Sizes Learns Sparse Features","version":2},"cited_work":{"arxiv_id":"2210.05337","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.05337","snapshot_observed_at":"2026-08-07T12:48:49.450826Z","title":"SGD with Large Step Sizes Learns Sparse Features","venue":"cs.LG","work_id":"bbfc9c5b-bc5d-4969-b26d-76535d7b3967","year":2022},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:36.710873Z"},"links":{"cited_paper":"/paper/2210.05337","citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:db895f97de175e13a4449246026ef8acfabeffcb9a7a45e30290c167259a240f","observation_id":"3243b1bb-8435-4704-81a1-041a34cd44dd","resolution":{"observed_at":"2026-08-07T12:48:49.648252Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.11477","last_updated":"2020-10-22T06:09:10Z","snapshot_observed_at":"2026-08-14T22:05:39.651975Z","submitted_at":"2020-06-20T02:35:02Z","title":"wav2vec 2.0: A Framework for Self-Supervised Learning of Speech Representations","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.11477","snapshot_observed_at":"2026-08-07T12:48:36.852990Z","title":"wav2vec 2.0: A framework for self-supervised learning of speech representations","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:36.852990Z"},"links":{"cited_paper":"/paper/2006.11477","citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:98d56b8a5d3ca34cd5d6337c61528a0b5f352a80fb73618d8cbf56ff78e8489c","observation_id":"517aaa9f-c710-4dc6-b5a0-076d38697fad","resolution":{"observed_at":"2026-08-07T12:48:36.852990Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:49:02.284823Z","title":"Implicit gradient regularization","venue":null,"work_id":"928e84d2-13a6-4dfc-a9a9-90b7b76f0b24","year":2021},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:36.995113Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:e765aa9d2f61ad68ac6059ecd53fe38092b56013be348f99709b43cd36fa8fe0","observation_id":"2fdbb3af-20d8-42ed-897c-9cb699d81e6e","resolution":{"observed_at":"2026-08-07T12:49:02.405254Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:49:02.069019Z","title":"Reconciling modern machine- learning practice and the classical bias–variance trade-off.Proceedings of the National Academy of Science, 116(32):15849–15854, 2019","venue":null,"work_id":"9a3d193e-c027-4c64-bc5d-1e893494c6a0","year":2019},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:37.146064Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:9d09965d32a3334a69fffda1fe2ef269349ae98624c65e1170aab6c0ab21c5f2","observation_id":"6b82d6ee-f178-472c-ba2f-663d27710d13","resolution":{"observed_at":"2026-08-07T12:49:02.156661Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:37.300545Z","title":"Practical recommendations for gradient-based training of deep architectures","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:37.300545Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:d4aca5c339d864921699161513d1fb9899d759544f144c7f2ec29c576a68a32b","observation_id":"288f3599-7164-4196-a634-048bf1de99ce","resolution":{"observed_at":"2026-08-07T12:48:37.300545Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2005.14165","last_updated":"2020-07-22T19:47:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-05-28T17:29:03Z","title":"Language Models are Few-Shot Learners","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.14165","snapshot_observed_at":"2026-08-07T12:48:37.430097Z","title":"Language models are few-shot learners","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:37.430097Z"},"links":{"cited_paper":"/paper/2005.14165","citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:ad6dbd3ee10ff6fb74d01f5f3f5104ef230d8564f6070b0a90f39f72fb2364f3","observation_id":"fefda19d-bf62-476f-9342-a81d024e532b","resolution":{"observed_at":"2026-08-07T12:48:37.430097Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:49:01.799029Z","title":"Stochastic gradient descent performs variational infer- ence, converges to limit cycles for deep networks","venue":null,"work_id":"91902886-c8e8-44b6-b98f-6e3354f4f917","year":2018},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:37.613521Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:6a33011f647b141cdb400017ab373f4e0346dc63adcd0d5a2ddeca944dd16211","observation_id":"e3849048-9ae6-4931-9f0a-257e1a7a22f1","resolution":{"observed_at":"2026-08-07T12:49:01.941909Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:49:01.491348Z","title":"Entropy-SGD: Biasing gradient descent into wide valleys","venue":null,"work_id":"1f7669d6-f256-4c68-aec1-e2f5f2e30c7d","year":2017},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:37.787127Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:a5c1f02061836c2eb152913737a57565db38150561a4919ef86396b350d66f1a","observation_id":"fa6adc37-e9a1-4333-81b1-036b7b5ec6e9","resolution":{"observed_at":"2026-08-07T12:49:01.606523Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:49:01.231164Z","title":"Convergence diagnostics for stochastic gradient descent with constant learning rate","venue":null,"work_id":"a6cc3dab-73d3-4764-abf3-df2729530a94","year":2018},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:37.938749Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:522578f658eaaaf9e980706b24e9f28695f7f9f797fba3c96e05d8d3a2d38417","observation_id":"e3b06684-69c2-45c8-8162-1cc2a6d9f69b","resolution":{"observed_at":"2026-08-07T12:49:01.326095Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:49:00.924175Z","title":"Sudden drops in the loss: Syntax acquisition, phase transitions, and simplicity bias in MLMs","venue":null,"work_id":"9c943d3a-cb55-4bbe-b61a-2f122073a216","year":2024},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:38.046351Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:74516405cae1abeb65690712747c547d4738a03e1f93e08cde10109412504b6a","observation_id":"961ad5b2-8ec7-4673-bfc3-d1623dc07ab7","resolution":{"observed_at":"2026-08-07T12:49:01.066149Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:49:00.587951Z","title":"Stochastic collapse: How gra- dient noise attracts SGD dynamics towards simpler subnetworks","venue":null,"work_id":"0e0b6e90-d322-4ab7-984e-1be224eb458d","year":2023},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:38.125374Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:111aa33a5ac8156001ac4bea2cbc289eed7006eef64b1c0f933506e5ea06e1cb","observation_id":"cc138851-290c-4b68-8622-d741abb879fc","resolution":{"observed_at":"2026-08-07T12:49:00.734116Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:49:00.307738Z","title":"Symbolic discovery of optimization algorithms","venue":null,"work_id":"7dfa3230-ca8c-4383-8a5e-8d1716af3957","year":2023},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:38.200072Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:24b638f63ecd9e308c8ff66d03f067d3e05ad1eb3bf072e6e98d590830ed3e5b","observation_id":"7422fef2-f593-4b54-942f-20077021b893","resolution":{"observed_at":"2026-08-07T12:49:00.460505Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:49:00.015021Z","title":"Gradient descent on neural networks typically occurs at the edge of stability","venue":null,"work_id":"4f9be559-4547-40d5-890c-ab9ad8a75a71","year":2021},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:38.281967Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:5a394d6426256be492aa9adb1add7e54b5d22f0431ee0899cb6260c1d14b34a4","observation_id":"16d05eca-ae36-4b1d-8361-dbe6d71bc8d8","resolution":{"observed_at":"2026-08-07T12:49:00.159172Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:59.680235Z","title":"URL https://constructor.tech/products/ research-platform","venue":null,"work_id":"3eb5a9e2-53e9-42d9-8172-ebff82e4c06f","year":null},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:38.353729Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:d56ef8d695291ce05e8bf7da1838c7015fca811ab487bfea2667704a1d89b485","observation_id":"2b1a750f-4c4b-471b-ad38-a82be506ce70","resolution":{"observed_at":"2026-08-07T12:48:59.887943Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:59.491440Z","title":"Determining intrinsic dimension and entropy of high-dimensional shape spaces.Modeling and Simulation in Science, Engineering and Technology, pages 231–252,","venue":null,"work_id":"f4784121-a74b-4647-944e-7f14db0cc63e","year":null},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:38.448719Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:87e5f6f04ad333e7d83eba24ed4924d636c940a3d8764f7e5e4325d75cec0136","observation_id":"966fca58-2736-4112-81f1-87aba74e8ed3","resolution":{"observed_at":"2026-08-07T12:48:59.602661Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:59.252843Z","title":"Why do we need weight decay in modern deep learning? InAdvances in Neural Information Processing Systems, 2024","venue":null,"work_id":"d57cc9ef-2913-42e2-8b12-5fd89262cf63","year":2024},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:38.626357Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:7e7a45d94c0d85be3ba959982ff3cf396c69095cbffeffaab8ac3b0c1037d37b","observation_id":"1a6b142d-03bf-4d07-b6f5-f8c36e64625b","resolution":{"observed_at":"2026-08-07T12:48:59.338853Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:59.064252Z","title":"BERT: Pre-training of deep bidirectional transformers for language understanding","venue":null,"work_id":"3955e0c2-5a1c-4a75-be0f-9a124bf1dd4c","year":null},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:38.720001Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:9fa7278c28404591d35d046bb642d3b3a14f3ca4dab46026cea83f2f3f9e2af9","observation_id":"91df84ab-a8df-416c-aae8-7be012c5531e","resolution":{"observed_at":"2026-08-07T12:48:59.129696Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:58.735878Z","title":"Essentially no barriers in neural network energy landscape","venue":null,"work_id":"771ec59a-453a-4c27-90cf-7d69b4c1a429","year":null},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:38.899024Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:f74728fadd46f34bd892b9e94cef71de077bbe02d87ddf0258bfb21e4b013893","observation_id":"0918aa1a-5731-47e4-8aa1-ed4a0a3bc1fa","resolution":{"observed_at":"2026-08-07T12:48:58.803625Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:58.571709Z","title":"Du, Xiyu Zhai, Barnabas Poczos, and Aarti Singh","venue":null,"work_id":"a3f7c631-b98c-4f2f-95c3-bc3844ff87d6","year":2019},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:39.121524Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:160a473cc9a2529be3f66c299f58f00bde870fb363c7fa9f957544728ed6cc3e","observation_id":"a19d1065-d8b1-46db-8f53-43c663845821","resolution":{"observed_at":"2026-08-07T12:48:58.652430Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:58.346348Z","title":"A free-energy principle for representation learning","venue":null,"work_id":"e05239ee-7246-4fa0-9bbb-12ce2193daba","year":2020},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:39.323634Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:28751669d4e5b17d7021ec079b59cc4e24e00ff343514d8e07f1927676caed0c","observation_id":"76675590-5794-4a6a-8824-09adfc9074bd","resolution":{"observed_at":"2026-08-07T12:48:58.431590Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2020.30014","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:49.216742Z","title":"Fixed-time stable gradient flows: Applications to continuous- time optimization.IEEE Transactions on Automatic Control, 66(5):2002–2015, 2021","venue":null,"work_id":"b6b662a3-b0e6-493e-a752-1ac2a9283282","year":2002},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:39.510629Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:21be98952f885d6ba89c707520fda8bd4d1e25b31d7822376832d2fdaf31efce","observation_id":"d8a40732-1810-4aba-8157-d1c6b296f298","resolution":{"observed_at":"2026-08-07T12:48:49.309456Z","resolver_source":"raw_fallback","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1802.10026","last_updated":"2018-10-30T11:39:49Z","snapshot_observed_at":"2026-08-14T19:41:32.557843Z","submitted_at":"2018-02-27T17:13:28Z","title":"Loss Surfaces, Mode Connectivity, and Fast Ensembling of DNNs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1802.10026","snapshot_observed_at":"2026-08-07T12:48:39.612467Z","title":"Loss surfaces, mode connectivity, and fast ensembling of DNNs.Advances in Neural Information Processing Systems, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:39.612467Z"},"links":{"cited_paper":"/paper/1802.10026","citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:7a9074501521751b3ecea6113dc1191962057b7e28a0c7d84da81eea8aef83d8","observation_id":"8b483a68-5559-4e66-8690-821feaafcd8d","resolution":{"observed_at":"2026-08-07T12:48:39.612467Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:58.036594Z","title":"Stochastic training is not necessary for generalization","venue":null,"work_id":"3f88ca45-f2e4-4ca9-817e-4860b9b66a44","year":2022},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:39.777083Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:6fdb93c15b73c70d555b824f3b0da5f8f17f5561c0c00a533390cde933bcf0c9","observation_id":"576d4845-946c-43f6-86f6-1da01bcffbb9","resolution":{"observed_at":"2026-08-07T12:48:58.252193Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:57.750988Z","title":"Abrupt learning in transformers: A case study on matrix completion","venue":null,"work_id":"f82f6b17-a76f-48b8-8b96-e8c2f7620aaa","year":2024},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:39.902009Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:7c786afb568feae6f22fc1cae1c99e07369585e9c095bbb7c65668f750ec9ffc","observation_id":"e52ca576-43c7-40b2-81f5-4ab1eecbe448","resolution":{"observed_at":"2026-08-07T12:48:57.896204Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1512.03385","last_updated":"2015-12-10T19:51:55Z","snapshot_observed_at":"2026-07-06T04:39:28.429064Z","submitted_at":"2015-12-10T19:51:55Z","title":"Deep Residual Learning for Image Recognition","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1512.03385","snapshot_observed_at":"2026-08-07T12:48:39.993026Z","title":"Deep residual learning for image recognition.Conference on Computer Vision and Pattern Recognition, 2016","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:39.993026Z"},"links":{"cited_paper":"/paper/1512.03385","citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:8209c5b1330174274500e2eb925ca878a707dcea2b5154c8ae41ed4670d111e9","observation_id":"93369438-8a66-4d8b-96bc-4e3ecda53feb","resolution":{"observed_at":"2026-08-07T12:48:39.993026Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1711.04623","last_updated":"2018-09-13T09:29:55Z","snapshot_observed_at":"2026-08-14T20:14:15.238615Z","submitted_at":"2017-11-13T15:11:56Z","title":"Three Factors Influencing Minima in SGD","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.04623","snapshot_observed_at":"2026-08-07T12:48:40.106339Z","title":"Three factors influencing minima in SGD","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:40.106339Z"},"links":{"cited_paper":"/paper/1711.04623","citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:f6539911b97fa77d85fefc7815d9cfae3207032328c1ca6b95b0a828b764ae88","observation_id":"73431ffa-d71a-4c42-8e26-5b6a9fba3ed9","resolution":{"observed_at":"2026-08-07T12:48:40.106339Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2001.08361","last_updated":"2020-01-23T03:59:20Z","snapshot_observed_at":"2026-08-13T17:41:53.092611Z","submitted_at":"2020-01-23T03:59:20Z","title":"Scaling Laws for Neural Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2001.08361","snapshot_observed_at":"2026-08-07T12:48:40.245350Z","title":"Brown, Benjamin Chess, Rewon Child, Scott Gray, Alec Radford, Jeffrey Wu, and Dario Amodei","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:40.245350Z"},"links":{"cited_paper":"/paper/2001.08361","citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:93289ffd3a869b6565149f859a15865f1fbb7f11e668f676ba33fd642c32714f","observation_id":"25c204db-ca40-4720-a820-5da35deaa3fd","resolution":{"observed_at":"2026-08-07T12:48:40.245350Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:57.486597Z","title":"Kingma and Jimmy Ba","venue":null,"work_id":"13c2c78a-c430-4b1b-857e-4300d64f9753","year":2015},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:40.382897Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:0cb023df1a2499e3af9b56097bf27dc5e4349b57de6dc1ea36e36a21ec267936","observation_id":"2015c824-c544-4b57-b3cf-11d281e975b5","resolution":{"observed_at":"2026-08-07T12:48:57.653712Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:57.185906Z","title":"Training scale-invariant neural networks on the sphere can happen in three regimes","venue":null,"work_id":"23addfd5-0112-497c-935c-b9108b6f5f8c","year":2022},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:40.449755Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:dad777c145f13ce4e8d0b4ea7df0a8fd093acda94e5ede759dd12fa27a016e2c","observation_id":"231877f2-6a5b-4758-aa3a-d070439aed62","resolution":{"observed_at":"2026-08-07T12:48:57.320267Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1912.11370","last_updated":"2020-05-05T20:48:23Z","snapshot_observed_at":"2026-08-14T13:34:41.782416Z","submitted_at":"2019-12-24T14:04:11Z","title":"Big Transfer (BiT): General Visual Representation Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1912.11370","snapshot_observed_at":"2026-08-07T12:48:40.523882Z","title":"Big transfer (BiT): General visual representation learning","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:40.523882Z"},"links":{"cited_paper":"/paper/1912.11370","citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:b15b2c56302b9862c053aab7b12ed62c5bc61a891138cdb28a2b395617eaf9fd","observation_id":"4d1962e9-77e7-4320-8ddd-65a345c5c553","resolution":{"observed_at":"2026-08-07T12:48:40.523882Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:56.859284Z","title":"CIFAR-10 (canadian institute for advanced research)","venue":null,"work_id":"7655fc27-44c1-46a3-ab02-791fc3f0c363","year":null},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:40.657045Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:d503ffe1dc70d9b3d579752682c4a112218a52466a97244fc5da7bb5cf83fe5a","observation_id":"eee4deaa-30ca-43ae-a0b2-0d42a487b4d4","resolution":{"observed_at":"2026-08-07T12:48:57.030510Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:56.604352Z","title":"CIFAR-100 (canadian institute for advanced research)","venue":null,"work_id":"4e1d9f2e-ee84-4f42-b9be-b2f677733b30","year":null},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:40.746665Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:ec27dd5e45949f8af03f30a5d3afd932ed1b067b3fe27a2488433b7fe258a273","observation_id":"32b601c5-e9d5-4cd7-9a81-e58de55d9de6","resolution":{"observed_at":"2026-08-07T12:48:56.712743Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1162/neco_a_01626","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":null,"venue":"Neural Computation","work_id":"76c1c20e-0493-4c02-8a31-2f216a7fa986","year":2023},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:40.831671Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:ad96b331d44b790caaad71692c970feb3a831b73ae0287853df4dafc19618a17","observation_id":"19d3eb95-9c45-4b8a-be47-56a9d7b1d7f9","resolution":{"observed_at":"2026-08-07T12:48:47.794906Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1907.04595","last_updated":"2020-04-25T06:30:13Z","snapshot_observed_at":"2026-08-14T16:07:10.415816Z","submitted_at":"2019-07-10T09:47:43Z","title":"Towards Explaining the Regularization Effect of Initial Large Learning Rate in Training Neural Networks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1907.04595","snapshot_observed_at":"2026-08-07T12:48:40.958072Z","title":"Towards explaining the regularization effect of initial large learning rate in training neural networks.Advances in Neural Information Processing Systems, 32, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:40.958072Z"},"links":{"cited_paper":"/paper/1907.04595","citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:86090f0f467c09d17934a23c69bef8501d14a85174589cd45a93e905619ccf76","observation_id":"e8891303-7c11-46e5-a72a-ea91e4b653c2","resolution":{"observed_at":"2026-08-07T12:48:40.958072Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1007/s10462-024-10915-y","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T00:03:40.895421Z","title":"Few-shot adaptation of multi-modal foundation models: A survey.Artificial Intelli- gence Review, 57(10):268, 2024","venue":"Artificial Intelligence Review","work_id":"26397c64-f172-429d-8b02-fcdbaf7c289e","year":2024},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:41.143952Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:394253fca9765205de8db0ae284b4cbbcffa4ee95c538e31c8e918b2bb753ace","observation_id":"ca1d1278-3582-4bfe-94dc-f0c1cdf38cfe","resolution":{"observed_at":"2026-08-07T12:48:47.484013Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:56.363713Z","title":"Understanding why neural networks generalize well through GSNR of parameters","venue":null,"work_id":"7f6519f8-5ea8-48e0-ae52-77502ae4984c","year":2020},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:41.225917Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:3b1e71b01149ef28bb5ed5b528f198c0b48be08cd211a8d5b80e2eb0a0bd7626","observation_id":"f575c99c-1242-4d96-af71-e05728df534e","resolution":{"observed_at":"2026-08-07T12:48:56.450939Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2012.03636","last_updated":"2021-06-11T08:31:26Z","snapshot_observed_at":"2026-08-13T20:50:23.110688Z","submitted_at":"2020-12-07T12:31:43Z","title":"Noise and Fluctuation of Finite Learning Rate Stochastic Gradient Descent","version":4},"cited_work":{"arxiv_id":"2012.03636","doi":null,"metadata_source":"pith","pith_arxiv_id":"2012.03636","snapshot_observed_at":"2026-08-07T12:48:48.801575Z","title":"Noise and Fluctuation of Finite Learning Rate Stochastic Gradient Descent","venue":"stat.ML","work_id":"a058990e-b297-4628-acf1-d14b3ce78fff","year":2020},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:41.344996Z"},"links":{"cited_paper":"/paper/2012.03636","citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:a86b8f9ecef6d514e6411f1d670a768a18c521b91995585d0343f292f3559441","observation_id":"8241ca34-06a4-40c1-914d-6141f0766561","resolution":{"observed_at":"2026-08-07T12:48:48.941810Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:56.111383Z","title":"Towards understanding grokking: An effective theory of representation learning","venue":null,"work_id":"8ad6d448-175b-47c4-a2d8-39cd51dcedd9","year":2022},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:41.465642Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:e75f803fe0c5f6cf3a7879994fff7c9075df42b6c3a8e26f273e92a4405d3da8","observation_id":"7c6e4e48-6a9e-4715-b430-90b2723e4a71","resolution":{"observed_at":"2026-08-07T12:48:56.234290Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:55.964195Z","title":"On the periodic behavior of neural network training with batch normalization and weight decay","venue":null,"work_id":"8b89fd71-4344-4988-b138-fe0a95223c17","year":2021},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:41.589945Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:79a9b0593db01821145bd76618ae7da5dcfab4e265a2a0adcf55eaba05d5bb1a","observation_id":"827a797e-dcfe-4010-befe-06364c2ae4b7","resolution":{"observed_at":"2026-08-07T12:48:56.036199Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:41.701378Z","title":"Decoupled weight decay regularization","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:41.701378Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:a55948e91d0e443e8713c7d42c5a0daf6b1290bf57d485a6f435f2aeaff10fe7","observation_id":"ebebddeb-4929-498a-92c9-45e220a39e4c","resolution":{"observed_at":"2026-08-07T12:48:41.701378Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.06559","last_updated":"2018-06-14T18:07:01Z","snapshot_observed_at":"2026-08-14T20:02:29.612441Z","submitted_at":"2017-12-18T18:10:39Z","title":"The Power of Interpolation: Understanding the Effectiveness of SGD in Modern Over-parametrized Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.06559","snapshot_observed_at":"2026-08-07T12:48:41.855492Z","title":"The power of interpolation: Understanding the effectiveness of SGD in modern over-parametrized learning","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:41.855492Z"},"links":{"cited_paper":"/paper/1712.06559","citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:e197d4bed1b9df7dd44457482fa490cf53bbf82a476746ca609783353a8b2c73","observation_id":"828f5cd0-057f-4141-a457-5f9059fdb4be","resolution":{"observed_at":"2026-08-07T12:48:41.855492Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1704.04289","last_updated":"2018-01-19T21:07:09Z","snapshot_observed_at":"2026-08-14T21:06:48.275336Z","submitted_at":"2017-04-13T22:17:30Z","title":"Stochastic Gradient Descent as Approximate Bayesian Inference","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1704.04289","snapshot_observed_at":"2026-08-07T12:48:41.985527Z","title":"Hoffman, and David M","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:41.985527Z"},"links":{"cited_paper":"/paper/1704.04289","citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:e18414f144e9cc3c09558ccf949a519fa16af80ef5c7d2e406eed3e5a0182790","observation_id":"77d1edcc-38ab-401e-b0ef-e3fa9d5b9094","resolution":{"observed_at":"2026-08-07T12:48:41.985527Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1088/2632-2153/ad1de6","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Phase transitions in the mini-batch size for sparse and dense two-layer neural networks.Machine Learning: Science and Technology, 5(1): 015015, 2024","venue":"Machine Learning Science and Technology","work_id":"2baf4c49-e1e7-4cc0-baa0-d8df8a76209d","year":2024},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:42.178037Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:72787791a22b221ea244e4c5f3e22cb5f886423d4f529e90901a3a6dfcc1e658","observation_id":"dbff6cb6-7569-4135-971f-733035b16bc0","resolution":{"observed_at":"2026-08-07T12:48:47.186730Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1806.01796","last_updated":"2022-04-18T14:12:57Z","snapshot_observed_at":"2026-08-14T19:07:14.576520Z","submitted_at":"2018-06-05T16:37:19Z","title":"Stochastic Gradient Descent on Separable Data: Exact Convergence with a Fixed Learning Rate","version":3},"cited_work":{"arxiv_id":"1806.01796","doi":null,"metadata_source":"pith","pith_arxiv_id":"1806.01796","snapshot_observed_at":"2026-08-07T12:48:48.536948Z","title":"Stochastic Gradient Descent on Separable Data: Exact Convergence with a Fixed Learning Rate","venue":"stat.ML","work_id":"b8b30ba2-ce1b-4c2b-8d27-25878e007127","year":2018},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:42.355392Z"},"links":{"cited_paper":"/paper/1806.01796","citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:658cd4169eec574ad43c1efd1c008e65f71f0f60061d12fb406246eca2fc1e1a","observation_id":"e3d4a475-5726-4fab-a6a2-aa39e708545e","resolution":{"observed_at":"2026-08-07T12:48:48.600944Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.15739","last_updated":"2023-04-20T05:35:31Z","snapshot_observed_at":"2026-08-13T12:15:08.047896Z","submitted_at":"2023-03-28T05:27:32Z","title":"Bayesian Free Energy of Deep ReLU Neural Network in Overparametrized Cases","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.15739","snapshot_observed_at":"2026-08-07T12:48:42.512102Z","title":"Bayesian free energy of deep ReLU neural network in overparametrized cases, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:42.512102Z"},"links":{"cited_paper":"/paper/2303.15739","citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:739db51e16789060602c2b178b658d8ccb7ddcc42582b08fa39afba994aaa559","observation_id":"c781f22f-abda-448a-8f2d-7ad84fdf099d","resolution":{"observed_at":"2026-08-07T12:48:42.512102Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:55.684171Z","title":"Deep double descent: Where bigger models and more data hurt","venue":null,"work_id":"387551e2-25e4-48dd-b3ba-42bd0b294e74","year":2020},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:42.607428Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:4a13651566e1ad594e93266d23c164fb7e862ba4a60d02c76c88454eb4863bdf","observation_id":"56031d6d-5de4-4f21-af93-d6af4d27f522","resolution":{"observed_at":"2026-08-07T12:48:55.828535Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:55.573203Z","title":"LR0.FM: Low-resolution zero-shot classification benchmark for foundation models","venue":null,"work_id":"c4016454-8f17-4f20-bd24-60ad5eda6cd0","year":2025},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:42.722384Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:4f3e0b2c130a80b583cd9c606ceaa2061c96658ee2246922fd4f365545fcc433","observation_id":"7ffc897f-72ea-47a6-bea6-e32e919e7dbe","resolution":{"observed_at":"2026-08-07T12:48:55.617982Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:55.517721Z","title":"Loss landscape: SGD has a better view","venue":null,"work_id":"30a3819e-3844-4704-b940-693654e11ec5","year":2020},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:42.817525Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:3310b1c23cec4c815b95eb70c9485368c0699e43828d993c2e5872f84677a121","observation_id":"1de94719-1f6a-4d71-bf51-888bc13cb8d8","resolution":{"observed_at":"2026-08-07T12:48:55.566771Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2201.02177","last_updated":"2022-01-06T18:43:37Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-01-06T18:43:37Z","title":"Grokking: Generalization Beyond Overfitting on Small Algorithmic Datasets","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2201.02177","snapshot_observed_at":"2026-08-07T12:48:42.923436Z","title":"Grokking: Generalization beyond overfitting on small algorithmic datasets, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:42.923436Z"},"links":{"cited_paper":"/paper/2201.02177","citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:57f19c4053cdf5729760f22e28ac7a7b87f3589718f51790bf84399b9ff52d3a","observation_id":"d099425f-38a9-4378-810d-088fb2301e2c","resolution":{"observed_at":"2026-08-07T12:48:42.923436Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.13681","last_updated":"2023-09-24T16:08:21Z","snapshot_observed_at":"2026-08-13T10:06:03.575320Z","submitted_at":"2023-09-24T16:08:21Z","title":"Accelerating Large Batch Training via Gradient Signal to Noise Ratio (GSNR)","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.13681","snapshot_observed_at":"2026-08-07T12:48:43.019137Z","title":"Accelerating large batch training via gradient signal to noise ratio (GSNR), 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:43.019137Z"},"links":{"cited_paper":"/paper/2309.13681","citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:d416987a9a6d6fbf545e41c824e0ed1a9950c05c8e1e12a381736990cd54ec65","observation_id":"37f347c5-f46a-48a5-9ea0-de6afe4c3d50","resolution":{"observed_at":"2026-08-07T12:48:43.019137Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:55.195721Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":"db20aee1-1f83-41a2-b954-c53f4dbbbb35","year":2021},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:43.098578Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:57c62eca0eb22507ee4ab17b6c54bb44edd86b8743b612dcfedccfd7c8df67f1","observation_id":"cf4970ed-0469-497f-8a72-61ba9cb8f5d3","resolution":{"observed_at":"2026-08-07T12:48:55.384360Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:55.008623Z","title":"Where do large learning rates lead us? InAdvances in Neural Information Processing Systems, 2024","venue":null,"work_id":"2c205118-6e60-4ded-8a29-a55533aa25b2","year":2024},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:43.211101Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:1a56f34fc426b5ed505c3502cb1184707528c775f52a901374ded60b2faf773f","observation_id":"c01f840f-b890-48a8-9bd6-8d601ae36896","resolution":{"observed_at":"2026-08-07T12:48:55.111425Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1073/pnas.2316301121","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"On the different regimes of stochastic gradient descent","venue":"Proceedings of the National Academy of Sciences","work_id":"f1466a6c-1ba3-47ba-b5ed-1025a75a3ff3","year":2024},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:43.295654Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:a21d00e03e198cd73ffc3aab4e1179354264ae89b594dfe2743292dc0a09ea20","observation_id":"0e38b7c4-cdf5-48af-8f49-a64a26b5ae39","resolution":{"observed_at":"2026-08-07T12:48:46.818627Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:54.696136Z","title":null,"venue":null,"work_id":"f981b295-7226-474b-88e2-425f47e8ac07","year":2018},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:43.496452Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:71551fceb70409c61e5fdfcec61b443f719c196dbd14a8d94ba4e20413ef21e2","observation_id":"8bff367d-c3eb-4f2e-9e81-ef15edefd85c","resolution":{"observed_at":"2026-08-07T12:48:54.855090Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.15081","last_updated":"2020-06-26T16:18:54Z","snapshot_observed_at":"2026-07-06T09:33:07.888450Z","submitted_at":"2020-06-26T16:18:54Z","title":"On the Generalization Benefit of Noise in Stochastic Gradient Descent","version":1},"cited_work":{"arxiv_id":"2006.15081","doi":null,"metadata_source":"pith","pith_arxiv_id":"2006.15081","snapshot_observed_at":"2026-08-07T12:48:48.154226Z","title":"On the Generalization Benefit of Noise in Stochastic Gradient Descent","venue":"cs.LG","work_id":"360d2719-0a94-48a3-9d27-588d4b3f9c6a","year":2020},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:43.574615Z"},"links":{"cited_paper":"/paper/2006.15081","citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:9c14d0941d7f7943f14779647161dda5cc28d424dec30c3bb2026b5210ce5724","observation_id":"7d61f117-34a8-45cb-ae16-2c4881bc476d","resolution":{"observed_at":"2026-08-07T12:48:48.284766Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:54.420680Z","title":"On the origin of implicit regular- ization in stochastic gradient descent","venue":null,"work_id":"da5503ec-f222-452c-9e75-64e3d29e1beb","year":null},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:43.678452Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:c1ae4cd42d0ef301b60511424e0fb0dd894a52d1882c0dd44ac4b1b69b2aa0df","observation_id":"f6fec504-538a-4709-8a10-96d443fc7f7d","resolution":{"observed_at":"2026-08-07T12:48:54.580089Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:53.924992Z","title":"Beyond the imitation game: Quantifying and extrapolating the capabilities of language models.Transactions on Machine Learning Research, 2023","venue":null,"work_id":"ed22c4ac-ea1d-4508-bf29-a6c59b6f3dec","year":2023},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:43.941915Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:47348221f3ecdb545bb4f58c2ac3834e2d34e599dee3f45aed5a014fbf4528b8","observation_id":"53d40a8b-67e4-44b8-a71e-b3fa1970713e","resolution":{"observed_at":"2026-08-07T12:48:54.032448Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:44.069104Z","title":"Unleashing the power of gradient signal-to-noise ratio for zero-shot NAS","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:44.069104Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:ac100332ce5eed6a8aea3c5af53b48e8c9b9cf8548f4361801cfb524e8d0102d","observation_id":"f553db86-34a5-4eb8-bc03-012018789d0a","resolution":{"observed_at":"2026-08-07T12:48:44.069104Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:53.581128Z","title":"Deep learning and the information bottleneck principle,","venue":null,"work_id":"086d1f4c-d07f-4fc4-b386-9526374e1169","year":null},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:44.188196Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:bdc6e888dc93d108496161950a073f9210261bf581bfc9a749ea52e6eeb63e96","observation_id":"328c658d-3707-4117-8e0a-39fdf8de6db0","resolution":{"observed_at":"2026-08-07T12:48:53.747474Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:53.298881Z","title":"The discovery of superconductivity.Physics Today, 63(9):38–43,","venue":null,"work_id":"c2d48424-4773-43c9-8712-fc7350efa67d","year":null},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:44.366536Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:3786a9ae0d96f916a44fc040471c2601e9fbb297fc00038f3299c5fc4f2b2b0b","observation_id":"c6382461-7c03-4953-8b2a-bf1fc8f2f61e","resolution":{"observed_at":"2026-08-07T12:48:53.425800Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:52.978068Z","title":"A survey of basic thermodynamics, 2004","venue":null,"work_id":"4d924a6e-4594-4f6d-b9c3-dd2a714abd58","year":2004},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:44.586912Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:2e7dfcb061f783d0884ddff1f50db68a81e9f21d5c38ca164c6afc820bdea84b","observation_id":"c81ca0d2-a013-4afd-9237-66511ad32f92","resolution":{"observed_at":"2026-08-07T12:48:53.156648Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:52.669935Z","title":"Chi, Tatsunori Hashimoto, Oriol Vinyals, Percy Liang, Jeff Dean, and William Fedus","venue":null,"work_id":"b14d11ef-9f26-4034-b2e9-34967bd9718d","year":2022},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:44.741482Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:1cbe3246feb64c3ba060e1795e27ca232b504d27c6e6d3c480cc06d17dd5db19","observation_id":"a41a7fe5-488a-4fe5-99fa-cfca3b4be752","resolution":{"observed_at":"2026-08-07T12:48:52.817300Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:52.320220Z","title":"Bayesian learning via stochastic gradient langevin dynamics","venue":null,"work_id":"2b64f026-1275-4543-a53d-c417887f0333","year":2011},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:44.876386Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:14fd0e9ff4ac25c8b51680b649cc634c28ffecb8c8fb7d5a9f4dd0bb4f162646","observation_id":"6e7e1725-9700-4565-b8c1-8edda39592b0","resolution":{"observed_at":"2026-08-07T12:48:52.531469Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:51.967218Z","title":"Towards few- shot adaptation of foundation models via multitask finetuning","venue":null,"work_id":"1c999d2a-793e-4bf0-868c-e633c8d4a6ff","year":2024},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:44.979783Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:826efb75cc1a41dd2a7f39ad6cec9bc6c63ab6e7d9552bceb958312260afebc2","observation_id":"d38b50ff-745b-4cc9-a359-5708d3fc7698","resolution":{"observed_at":"2026-08-07T12:48:52.147253Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:51.632538Z","title":"Fluctuation-dissipation relations for stochastic gradient descent","venue":null,"work_id":"bf703d9f-00b6-4524-83f1-f40a5ac5ffbb","year":2019},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:45.148706Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:154f1644b24e8e8d16900cc83feb6227708ef43273aac760df702078bd11c3a4","observation_id":"fc4b5a34-80a4-473f-af50-29d8dd8d6472","resolution":{"observed_at":"2026-08-07T12:48:51.794687Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.01878","last_updated":"2019-09-26T06:53:59Z","snapshot_observed_at":"2026-08-14T17:18:02.477401Z","submitted_at":"2019-08-05T21:56:41Z","title":"How Does Learning Rate Decay Help Modern Neural Networks?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.01878","snapshot_observed_at":"2026-08-07T12:48:45.281153Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:45.281153Z"},"links":{"cited_paper":"/paper/1908.01878","citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:a36101c1f0096eb078a121d5eb8d4a90166331adc33b128168a69dc6d577e2ef","observation_id":"faf4f39e-dd6a-4079-a3d0-c9e1dd180e17","resolution":{"observed_at":"2026-08-07T12:48:45.281153Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:45.411715Z","title":"Saxe, Madhu S","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:45.411715Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:0233ca3bdfdda43a66f713d1a086b99bbab61d7486bd77e3872222f8fc9d3bb5","observation_id":"7b3fab21-76e1-4d7e-ba52-3c684d880759","resolution":{"observed_at":"2026-08-07T12:48:45.411715Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:51.201929Z","title":"Strength of minibatch noise in SGD","venue":null,"work_id":"e8f48749-d771-440d-bd68-53d41e69498c","year":2022},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:45.569826Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:c4bfffaf342540a9ee9a4da8e2656789904a601b1046a1bafeda478b2fc3fb23","observation_id":"81a29911-f563-4fdb-99d5-79ca2eeb6d7f","resolution":{"observed_at":"2026-08-07T12:48:51.465143Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:50.923597Z","title":"Stochastic gradient descent opti- mizes over-parameterized deep ReLU networks, 2018","venue":null,"work_id":"c904b92b-8899-4218-82af-20df50f26aa1","year":2018},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:45.709225Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:0b649b55057960c1b41c37ff155210a956d9b1532c6a9f18e6b79122cc0cd5b3","observation_id":"23174278-5a2e-4102-a24b-ad149a96c37a","resolution":{"observed_at":"2026-08-07T12:48:51.044331Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:50.715557Z","title":"g., unit sphere in our case), can be interpreted as fixed volume","venue":null,"work_id":"53f7dfc0-0881-42ee-98cb-e2fb972bd454","year":null},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:45.857082Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:45fb4d48434cf20b4bd1275fbd030995c5e20d99c80ddf9b8bf5455df88fc442","observation_id":"81c054d8-a7b3-4c13-be06-505ae14b05ce","resolution":{"observed_at":"2026-08-07T12:48:50.794185Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:50.463179Z","title":null,"venue":null,"work_id":"1671720b-1285-4d53-a27c-1f81a2d36a87","year":null},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:45.936357Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:7b4da27a8911598954356e515a58e24695e0b8d0deff3665c14456f95e75ba0d","observation_id":"93744545-3594-4ef4-80f8-fe9de8266bcb","resolution":{"observed_at":"2026-08-07T12:48:50.601439Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:50.150916Z","title":"An additional justification for using the Helmholtz free energy arises from the stationary distributions of SGD","venue":null,"work_id":"edbfca35-b0c9-4b8f-b7a4-7800fd7d5b72","year":null},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:46.143996Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:ee45d80a1fb46295ea7954310d08456bdd957a5b5df7c80fa44715c15797a45a","observation_id":"72fb72a9-53b7-49f6-8bd8-c24002a51a42","resolution":{"observed_at":"2026-08-07T12:48:50.308842Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:38.550624Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":2007,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:38.550624Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:347183251096ee93452c9ba7dac702d83e3982af310276ce323563c4615f3d8f","observation_id":"2e36e8c0-bfb9-47c7-831e-035b7aa6f484","resolution":{"observed_at":"2026-08-07T12:48:38.550624Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1063/1.3490499","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":null,"venue":"Physics Today","work_id":"26bbffca-6c57-44e7-871d-dcd5e5cadc14","year":null},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":2010,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:44.474011Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:08ba996bfed5caad2f75f088b4c7f83b2e59bd33b839a0ac120b6c774c2906c1","observation_id":"11e721c2-df99-4bf3-a9f3-95e2df384f1c","resolution":{"observed_at":"2026-08-07T12:48:46.474308Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1503.02406","last_updated":"2015-03-09T09:39:41Z","snapshot_observed_at":"2026-08-14T22:57:12.242889Z","submitted_at":"2015-03-09T09:39:41Z","title":"Deep Learning and the Information Bottleneck Principle","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1503.02406","snapshot_observed_at":"2026-08-07T12:48:44.291906Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":2015,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:44.291906Z"},"links":{"cited_paper":"/paper/1503.02406","citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:84558dd3b2c6a22aa20fd9330bc7a47ead73e42323441282722a95ba542be6f8","observation_id":"8b7de796-ebd5-4559-89b7-2c70f9b31ada","resolution":{"observed_at":"2026-08-07T12:48:44.291906Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1803.00885","last_updated":"2019-02-22T11:20:22Z","snapshot_observed_at":"2026-08-14T19:40:15.802272Z","submitted_at":"2018-03-02T15:22:10Z","title":"Essentially No Barriers in Neural Network Energy Landscape","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1803.00885","snapshot_observed_at":"2026-08-07T12:48:38.980064Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":2018,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:38.980064Z"},"links":{"cited_paper":"/paper/1803.00885","citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:7199a4ff4c6097d5b1bc7a2b15b43c039e1acc1f76318a5b99f5ffc977f22c79","observation_id":"fc848c5b-e058-4feb-9c30-8229cef5be8e","resolution":{"observed_at":"2026-08-07T12:48:38.980064Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:58.896992Z","title":null,"venue":null,"work_id":"f53f1f56-9670-4eb8-ac97-c14b1d1b2f6e","year":null},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:38.809372Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:efb11a7343072d201a27b6086d9e482b4d051f50b7e365a7e04085e9ea067d30","observation_id":"541da72f-9f5e-4f2a-b0ea-81eb4a9d7495","resolution":{"observed_at":"2026-08-07T12:48:58.968877Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:54.110293Z","title":null,"venue":null,"work_id":"0d577865-03d7-494f-8ec4-54344ccff9d0","year":null},"citing_paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:43.792243Z"},"links":{"citing_paper":"/paper/2505.23489"},"observation_digest":"sha256:6e908c51650a82eb2705e56880ed78aabebcab71b52e4843c4d3bbdd91b47559","observation_id":"ed43d29b-cfc1-4ed2-a31a-9f7e8047b64f","resolution":{"observed_at":"2026-08-07T12:48:54.317558Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.23489","last_updated":"2025-05-29T14:40:24Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-14T22:06:26.866660Z","submitted_at":"2025-05-29T14:40:24Z","title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training"},"reference_resolution":{"displayed":79,"state_counts":{"malformed_identifier":1,"metadata_mismatch":3,"parse_uncertain":0,"unresolved":25,"verified_exact":8,"verified_fuzzy":42},"total_outbound_references":79},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 79 of 79 outbound references and 0 inbound Pith citation observations for arXiv:2505.23489."}