{"as_of":"2026-08-09T06:37:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e42ba70bb0984d2559cb983093ee118117c568e0b74d5547ac7c42138825654e","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":37,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":37,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":37,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":37,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T12:12:47.569722Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":1,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-07T10:31:20.534009Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05447","last_updated":"2025-07-14T23:29:38Z","snapshot_observed_at":"2026-08-09T06:27:45.424360Z","submitted_at":"2025-06-05T15:18:35Z","title":"Training Dynamics Underlying Language Model Scaling Laws: Loss Deceleration and Zero-Sum Learning","version":2},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-07T10:31:20.534009Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2506.05447"},"observation_digest":"sha256:9211eddfc7188ceddcbbb0dc80217a3a1280974e94dbb24c172d343228c4e331","observation_id":"22a06aaa-38ec-4f17-98a5-881b1a0aa397","resolution":{"observed_at":"2026-08-07T10:31:20.534009Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-07T12:12:47.569722Z","title":"Understanding warmup-stable-decay learning rates: A river valley loss landscape perspective, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.08027","last_updated":"2025-08-18T19:51:06Z","snapshot_observed_at":"2026-08-08T01:05:08.380869Z","submitted_at":"2025-05-30T21:08:15Z","title":"Recipes for Pre-training LLMs with MXFP8","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T12:12:47.569722Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2506.08027"},"observation_digest":"sha256:4f512169e3f459f0a2f702356444698114dfcf4f76228ca52c702b6c04d26cdd","observation_id":"cefd3c8d-4ce7-45c9-99ff-acb5919ea260","resolution":{"observed_at":"2026-08-07T12:12:47.569722Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-07T04:21:45.283461Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10896","last_updated":"2025-06-12T17:01:11Z","snapshot_observed_at":"2026-08-07T20:56:34.872687Z","submitted_at":"2025-06-12T17:01:11Z","title":"BioClinical ModernBERT: A State-of-the-Art Long-Context Encoder for Biomedical and Clinical NLP","version":1},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-07T04:21:45.283461Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2506.10896"},"observation_digest":"sha256:5afb256819b6115e2450b5f9839931304078b2151ca385df8f08131fbf3624b0","observation_id":"b5313255-eb34-44b3-a3b9-984f3cf42d19","resolution":{"observed_at":"2026-08-07T04:21:45.283461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-06T23:20:55.558998Z","title":"Understanding warmup-stable-decay learning rates: A river valley loss landscape perspective","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.18639","last_updated":"2025-06-23T13:42:00Z","snapshot_observed_at":"2026-08-08T14:58:53.632163Z","submitted_at":"2025-06-23T13:42:00Z","title":"ByteSpan: Information-Driven Subword Tokenisation","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-06T23:20:55.558998Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2506.18639"},"observation_digest":"sha256:0108dbc2cde2ccebc1aaa0e929ab4f347a84ab5fea9a813fec5fcc8046f483cb","observation_id":"c175f1fc-bcb7-4348-ade2-8c41d48d21d6","resolution":{"observed_at":"2026-08-06T23:20:55.558998Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-06T20:48:54.682273Z","title":"Understanding warmup-stable-decay learning rates: A river valley loss landscape perspective","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02119","last_updated":"2025-07-07T06:13:26Z","snapshot_observed_at":"2026-08-09T03:48:37.458741Z","submitted_at":"2025-07-02T20:03:34Z","title":"Scaling Collapse Reveals Universal Dynamics in Compute-Optimally Trained Neural Networks","version":2},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-06T20:48:54.682273Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2507.02119"},"observation_digest":"sha256:9c2055d52fd8f7323c94ba701934995ad93e00987615e0902f932cc5a6518b70","observation_id":"8e56a2c6-68f4-468a-b2d3-e557308434b4","resolution":{"observed_at":"2026-08-06T20:48:54.682273Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-06T19:59:05.216957Z","title":"Understanding warmup-stable- decay learning rates: A river valley loss landscape perspective","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.04206","last_updated":"2025-07-06T01:34:12Z","snapshot_observed_at":"2026-08-06T19:50:42.879569Z","submitted_at":"2025-07-06T01:34:12Z","title":"Mpemba Effect in Large-Language Model Training Dynamics: A Minimal Analysis of the Valley-River model","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T19:59:05.216957Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2507.04206"},"observation_digest":"sha256:3872992d97351e4e16f81b33a7f57b6803390d5a96bc874cc9bfc762598ef63b","observation_id":"46ec7f39-826e-43f3-ad28-e178f90c2d64","resolution":{"observed_at":"2026-08-06T19:59:05.216957Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-06T14:49:40.357950Z","title":"Understanding warmup-stable-decay learning rates: A river valley loss landscape perspective","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17634","last_updated":"2025-08-11T08:36:31Z","snapshot_observed_at":"2026-08-06T19:16:35.269127Z","submitted_at":"2025-07-23T16:02:06Z","title":"WSM: Decay-Free Learning Rate Schedule via Checkpoint Merging for LLM Pre-training","version":2},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-06T14:49:40.357950Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2507.17634"},"observation_digest":"sha256:b7dac657315a94a4579350856d7cff444e9116f0e351f828e78b779a549bb1a0","observation_id":"448880cd-8952-4b5d-af44-e21859de42a6","resolution":{"observed_at":"2026-08-06T14:49:40.357950Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-04T21:32:35.801321Z","title":"Understanding warmup-stable-decay learning rates: A river valley loss landscape perspective","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.07972","last_updated":"2025-09-09T17:56:03Z","snapshot_observed_at":"2026-08-04T21:32:29.559391Z","submitted_at":"2025-09-09T17:56:03Z","title":"Theoretical Analysis on how Learning Rate Warmup Accelerates Convergence","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-04T21:32:35.801321Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2509.07972"},"observation_digest":"sha256:fd6d259fc03e02ebb18727444b1458aac3b0c6354fb83c887006ed959fa123c6","observation_id":"3e405299-3eb1-4fca-9f5b-0bc3331cc7b8","resolution":{"observed_at":"2026-08-04T21:32:35.801321Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-04T20:07:03.143145Z","title":"Wang, David Hall, Percy Liang, and Tengyu Ma","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09737","last_updated":"2025-09-10T18:01:04Z","snapshot_observed_at":"2026-08-08T13:38:01.687801Z","submitted_at":"2025-09-10T18:01:04Z","title":"World Modeling with Probabilistic Structure Integration","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-04T20:07:03.143145Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2509.09737"},"observation_digest":"sha256:be696b4d2233dc2a759fe40f40f61a7997795cd027526b3b89443e3ed5e794f3","observation_id":"1c5a5312-e54b-4b57-8b03-ebbf370efd63","resolution":{"observed_at":"2026-08-04T20:07:03.143145Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-04T12:39:01.733820Z","title":"Understanding warmup-stable-decay learning rates: A river valley loss landscape perspective","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2510.03164","last_updated":"2026-06-28T16:14:54Z","snapshot_observed_at":"2026-08-04T12:38:48.240318Z","submitted_at":"2025-10-03T16:35:56Z","title":"Why Do We Need Warm-up? A Theoretical Perspective","version":2},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-04T12:39:01.733820Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2510.03164"},"observation_digest":"sha256:7402cd86ce6fd724d646dd113d35d6774c8c0f2b1dae5518614b21ef5267e178","observation_id":"bc431704-a053-41e8-993a-0db5dd316058","resolution":{"observed_at":"2026-08-04T12:39:01.733820Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":"2410.05192","doi":"10.48550/arxiv.2410.05192","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding warmup-stable-decay learning rates: A river valley loss landscape perspective","venue":"arXiv (Cornell University)","work_id":"a4ab5b82-d4c2-4e3c-9f77-9c9de5253875","year":2024},"citing_paper":{"arxiv_id":"2510.25741","last_updated":"2026-07-01T23:25:58Z","snapshot_observed_at":"2026-08-04T07:30:51.188041Z","submitted_at":"2025-10-29T17:45:42Z","title":"Scaling Latent Reasoning via Looped Language Models","version":4},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-15T07:43:11.620446Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2510.25741"},"observation_digest":"sha256:ab2b7408563bce1760fc22bb7e37abafb12ed25689e42eaf9bd39003de9fd5bf","observation_id":"de12a0e5-f26d-4e6c-8048-8af409992ab6","resolution":{"observed_at":"2026-05-15T07:43:11.863947Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-04T07:31:31.276350Z","title":"Understanding warmup-stable- decay learning rates: A river valley loss landscape perspective.arXiv preprint arXiv:2410.05192, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2510.25741","last_updated":"2026-07-01T23:25:58Z","snapshot_observed_at":"2026-08-04T07:30:51.188041Z","submitted_at":"2025-10-29T17:45:42Z","title":"Scaling Latent Reasoning via Looped Language Models","version":5},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-04T07:31:31.276350Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2510.25741"},"observation_digest":"sha256:15381849c2287779d176988afcf39d079c1e3e99db4d88c1cc68c10e4c68cb9d","observation_id":"29b67147-6a72-4e4d-afba-15310c7b5757","resolution":{"observed_at":"2026-08-04T07:31:31.276350Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-03T05:01:13.163417Z","title":"Understanding warmup-stable-decay learning rates: A river valley loss landscape perspective.arXiv preprint arXiv:2410.05192,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.03685","last_updated":"2026-05-31T19:36:07Z","snapshot_observed_at":"2026-08-03T05:01:07.831427Z","submitted_at":"2026-02-03T16:06:18Z","title":"Universal One-third Time Scaling in Learning Peaked Distributions","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-03T05:01:13.163417Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2602.03685"},"observation_digest":"sha256:ecdac38e43ffaacffce5798bd212d5e6d26b217f8832526b9e5550bd171188a6","observation_id":"7041df66-e14d-48fa-9b66-5f6612aa5bd6","resolution":{"observed_at":"2026-08-03T05:01:13.163417Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":"2410.05192","doi":"10.48550/arxiv.2410.05192","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding warmup-stable-decay learning rates: A river valley loss landscape perspective","venue":"arXiv (Cornell University)","work_id":"a4ab5b82-d4c2-4e3c-9f77-9c9de5253875","year":2024},"citing_paper":{"arxiv_id":"2602.04774","last_updated":"2026-05-08T16:24:57Z","snapshot_observed_at":"2026-08-03T01:34:12.578849Z","submitted_at":"2026-02-04T17:11:36Z","title":"Theory of Optimal Learning Rate Schedules and Scaling Laws for a Random Feature Model","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-16T06:58:38.927268Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2602.04774"},"observation_digest":"sha256:22b0ff1709d41c206b70b4b8fbe19325a37c9d39dcb49fcd4f8a08a604a757f4","observation_id":"2f493f42-1605-4fd8-9cc5-9a7417b627c6","resolution":{"observed_at":"2026-05-16T07:00:43.295910Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":"2410.05192","doi":"10.48550/arxiv.2410.05192","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding warmup-stable-decay learning rates: A river valley loss landscape perspective","venue":"arXiv (Cornell University)","work_id":"a4ab5b82-d4c2-4e3c-9f77-9c9de5253875","year":2024},"citing_paper":{"arxiv_id":"2604.18827","last_updated":"2026-04-20T20:46:46Z","snapshot_observed_at":"2026-07-06T23:05:35.374092Z","submitted_at":"2026-04-20T20:46:46Z","title":"OmniMouse: Scaling properties of multi-modal, multi-task Brain Models on 150B Neural Tokens","version":1},"reference_index":111,"source":"arxiv_source","source_observed_at":"2026-05-10T02:43:25.842048Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2604.18827"},"observation_digest":"sha256:d09839610c769ab06981db43c04b830e392fa81087945d70bcd7e138f70331b2","observation_id":"31e586fc-f1f2-4ca9-8473-b03c02c05ced","resolution":{"observed_at":"2026-05-11T12:51:04.770940Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":"2410.05192","doi":"10.48550/arxiv.2410.05192","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding warmup-stable-decay learning rates: A river valley loss landscape perspective","venue":"arXiv (Cornell University)","work_id":"a4ab5b82-d4c2-4e3c-9f77-9c9de5253875","year":2024},"citing_paper":{"arxiv_id":"2605.02105","last_updated":"2026-05-04T00:02:23Z","snapshot_observed_at":"2026-08-03T01:31:24.030995Z","submitted_at":"2026-05-04T00:02:23Z","title":"Sharpness-Aware Pretraining Mitigates Catastrophic Forgetting","version":1},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-05-08T18:33:26.638240Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2605.02105"},"observation_digest":"sha256:6e1dccd868142aedebf5c61f97edebd27c98b1b84d0e57b58db976c368a3af6f","observation_id":"683f7852-1c1c-4119-a4ce-c4a97b12c1cb","resolution":{"observed_at":"2026-05-09T06:20:42.358768Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":"2410.05192","doi":"10.48550/arxiv.2410.05192","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding warmup-stable-decay learning rates: A river valley loss landscape perspective","venue":"arXiv (Cornell University)","work_id":"a4ab5b82-d4c2-4e3c-9f77-9c9de5253875","year":2024},"citing_paper":{"arxiv_id":"2605.06366","last_updated":"2026-05-11T08:56:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-07T14:39:40Z","title":"Layer Collapse in Diffusion Language Models","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-08T12:52:12.104539Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2605.06366"},"observation_digest":"sha256:d5397346974d8bba45d61bec37f329c249d26b7c717f6f04bfd84c04a793898e","observation_id":"abfa34ad-b64a-409c-aa1e-1aba36e74d86","resolution":{"observed_at":"2026-05-11T19:01:18.918915Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":"2410.05192","doi":"10.48550/arxiv.2410.05192","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding warmup-stable-decay learning rates: A river valley loss landscape perspective","venue":"arXiv (Cornell University)","work_id":"a4ab5b82-d4c2-4e3c-9f77-9c9de5253875","year":2024},"citing_paper":{"arxiv_id":"2605.06366","last_updated":"2026-05-11T08:56:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-07T14:39:40Z","title":"Layer Collapse in Diffusion Language Models","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-12T03:13:32.889304Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2605.06366"},"observation_digest":"sha256:ba6ae34817e32a1eccdd3a462a77b18911e72928fb9f340b677aef290ec4dbeb","observation_id":"19d357ff-dfd8-4c89-81f8-7d56fc61ec68","resolution":{"observed_at":"2026-05-12T03:16:19.186092Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":"2410.05192","doi":"10.48550/arxiv.2410.05192","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding warmup-stable-decay learning rates: A river valley loss landscape perspective","venue":"arXiv (Cornell University)","work_id":"a4ab5b82-d4c2-4e3c-9f77-9c9de5253875","year":2024},"citing_paper":{"arxiv_id":"2605.06546","last_updated":"2026-05-19T10:02:00Z","snapshot_observed_at":"2026-07-06T23:19:00.794334Z","submitted_at":"2026-05-07T16:41:37Z","title":"Efficient Pre-Training with Token Superposition","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-08T10:09:56.187089Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2605.06546"},"observation_digest":"sha256:e9c1c21ee7e15568b8f22bef6bfabea5244018ac61648e6de7f26c42f46343e2","observation_id":"71ed0272-3d9e-4ec9-b888-e7a36989a4ba","resolution":{"observed_at":"2026-05-11T20:11:09.614732Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":"2410.05192","doi":"10.48550/arxiv.2410.05192","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding warmup-stable-decay learning rates: A river valley loss landscape perspective","venue":"arXiv (Cornell University)","work_id":"a4ab5b82-d4c2-4e3c-9f77-9c9de5253875","year":2024},"citing_paper":{"arxiv_id":"2605.06546","last_updated":"2026-05-19T10:02:00Z","snapshot_observed_at":"2026-07-06T23:19:00.794334Z","submitted_at":"2026-05-07T16:41:37Z","title":"Efficient Pre-Training with Token Superposition","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-20T22:54:12.913665Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2605.06546"},"observation_digest":"sha256:b0a9a0d64be83ed47404314e95b35a73846f10499b8dc2027564b19954bc083d","observation_id":"3ffed03c-4ffc-4387-a304-fb430ad2e4a6","resolution":{"observed_at":"2026-05-20T22:59:11.971847Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":"2410.05192","doi":"10.48550/arxiv.2410.05192","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding warmup-stable-decay learning rates: A river valley loss landscape perspective","venue":"arXiv (Cornell University)","work_id":"a4ab5b82-d4c2-4e3c-9f77-9c9de5253875","year":2024},"citing_paper":{"arxiv_id":"2605.13652","last_updated":"2026-05-19T00:27:00Z","snapshot_observed_at":"2026-07-06T23:25:11.623026Z","submitted_at":"2026-05-13T15:11:37Z","title":"Beyond Perplexity: A Geometric and Spectral Study of Low-Rank Pre-Training","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-14T19:19:18.275446Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2605.13652"},"observation_digest":"sha256:a5ae7f6353c9b8cc90874eff4e2f94869e90b7f2bfee83a5ce7edf6ebd9681f3","observation_id":"f7953c0e-5255-4547-aade-218bb215afe6","resolution":{"observed_at":"2026-05-14T19:19:23.577938Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":"2410.05192","doi":"10.48550/arxiv.2410.05192","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding warmup-stable-decay learning rates: A river valley loss landscape perspective","venue":"arXiv (Cornell University)","work_id":"a4ab5b82-d4c2-4e3c-9f77-9c9de5253875","year":2024},"citing_paper":{"arxiv_id":"2605.13652","last_updated":"2026-05-19T00:27:00Z","snapshot_observed_at":"2026-07-06T23:25:11.623026Z","submitted_at":"2026-05-13T15:11:37Z","title":"Beyond Perplexity: A Geometric and Spectral Study of Low-Rank Pre-Training","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-20T20:24:26.681383Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2605.13652"},"observation_digest":"sha256:ef96366e3731a684830b0f6fa01ba17768483a65b314fcb37f77dfd704e1a91e","observation_id":"25dcf6a0-b29f-4c60-a01b-df10affea01e","resolution":{"observed_at":"2026-05-20T20:28:59.913976Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":"2410.05192","doi":"10.48550/arxiv.2410.05192","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding warmup-stable-decay learning rates: A river valley loss landscape perspective","venue":"arXiv (Cornell University)","work_id":"a4ab5b82-d4c2-4e3c-9f77-9c9de5253875","year":2024},"citing_paper":{"arxiv_id":"2605.24956","last_updated":"2026-07-12T09:24:31Z","snapshot_observed_at":"2026-08-07T14:43:53.645603Z","submitted_at":"2026-05-24T09:13:12Z","title":"NITP: Next Implicit Token Prediction for LLM Pre-training","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-06-30T12:23:42.587689Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2605.24956"},"observation_digest":"sha256:446b36750d710730eb0858b4296ebf3188eef22329f5fc562014e9d079a5beeb","observation_id":"ce29ece4-1e27-467c-8ffe-d775327cc9b6","resolution":{"observed_at":"2026-06-30T12:24:39.258946Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":"2410.05192","doi":"10.48550/arxiv.2410.05192","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding warmup-stable-decay learning rates: A river valley loss landscape perspective","venue":"arXiv (Cornell University)","work_id":"a4ab5b82-d4c2-4e3c-9f77-9c9de5253875","year":2024},"citing_paper":{"arxiv_id":"2605.24956","last_updated":"2026-07-12T09:24:31Z","snapshot_observed_at":"2026-08-07T14:43:53.645603Z","submitted_at":"2026-05-24T09:13:12Z","title":"NITP: Next Implicit Token Prediction for LLM Pre-training","version":2},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-07-04T00:38:33.708836Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2605.24956"},"observation_digest":"sha256:ba7bb49dd10e0381029bf8005c133b458ddf9ebd62798232eee9539ec662e6b4","observation_id":"4fb95c48-2318-4f41-8e05-e6852da4db63","resolution":{"observed_at":"2026-07-04T00:39:16.247898Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-07-14T18:45:28.635910Z","title":"Understanding warmup-stable-decay learning rates: A river valley loss landscape perspective.arXiv preprint arXiv:2410.05192,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.24956","last_updated":"2026-07-12T09:24:31Z","snapshot_observed_at":"2026-08-07T14:43:53.645603Z","submitted_at":"2026-05-24T09:13:12Z","title":"NITP: Next Implicit Token Prediction for LLM Pre-training","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-07-14T18:45:28.635910Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2605.24956"},"observation_digest":"sha256:b5c6de59d4c431fe42bd20e860ce8b10da11df84a28188141dcf73ac2907b4a2","observation_id":"a1f1d358-8aa6-49fa-8bdb-b54f577e9629","resolution":{"observed_at":"2026-07-14T18:45:28.635910Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":"2410.05192","doi":"10.48550/arxiv.2410.05192","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding warmup-stable-decay learning rates: A river valley loss landscape perspective","venue":"arXiv (Cornell University)","work_id":"a4ab5b82-d4c2-4e3c-9f77-9c9de5253875","year":2024},"citing_paper":{"arxiv_id":"2606.00439","last_updated":"2026-05-30T00:10:32Z","snapshot_observed_at":"2026-08-02T22:20:28.613688Z","submitted_at":"2026-05-30T00:10:32Z","title":"Physical Object Understanding with a Physically Controllable World Model","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-06-28T19:31:32.409769Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2606.00439"},"observation_digest":"sha256:328b99816a117a59a00de5e3a287a5437291a27bd5ec2a9042231b92c2864ba8","observation_id":"38230947-1aa5-4bcb-a1c4-40a33a800e66","resolution":{"observed_at":"2026-06-28T19:32:34.737365Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":"2410.05192","doi":"10.48550/arxiv.2410.05192","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding warmup-stable-decay learning rates: A river valley loss landscape perspective","venue":"arXiv (Cornell University)","work_id":"a4ab5b82-d4c2-4e3c-9f77-9c9de5253875","year":2024},"citing_paper":{"arxiv_id":"2606.04327","last_updated":"2026-06-03T01:03:34Z","snapshot_observed_at":"2026-08-02T12:14:20.139049Z","submitted_at":"2026-06-03T01:03:34Z","title":"A Geometric Characterization of the Stationary Plateau for Two-Layer Neural Networks","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-06-28T07:25:06.419331Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2606.04327"},"observation_digest":"sha256:8b0120185d683de0ce8a9ab2b8927b3ef3d025db2cf94a7cba3cf5317bd52c55","observation_id":"9586509c-23fd-49fb-9486-ee5fb846db46","resolution":{"observed_at":"2026-07-02T06:16:44.014554Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":"2410.05192","doi":"10.48550/arxiv.2410.05192","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding warmup-stable-decay learning rates: A river valley loss landscape perspective","venue":"arXiv (Cornell University)","work_id":"a4ab5b82-d4c2-4e3c-9f77-9c9de5253875","year":2024},"citing_paper":{"arxiv_id":"2606.05610","last_updated":"2026-06-04T02:32:11Z","snapshot_observed_at":"2026-07-06T23:45:33.157370Z","submitted_at":"2026-06-04T02:32:11Z","title":"Predictable Scaling Laws of Optimal Hyperparameters for LLM Continued Pre-training","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-06-28T01:53:04.715108Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2606.05610"},"observation_digest":"sha256:ff88b8db01a0650e439e881e26f27982b180e41bc6cbf730e33b83fbe5464770","observation_id":"df13f7f3-3663-48d7-86d1-232aeb0b4fae","resolution":{"observed_at":"2026-07-02T12:46:56.662592Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":"2410.05192","doi":"10.48550/arxiv.2410.05192","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding warmup-stable-decay learning rates: A river valley loss landscape perspective","venue":"arXiv (Cornell University)","work_id":"a4ab5b82-d4c2-4e3c-9f77-9c9de5253875","year":2024},"citing_paper":{"arxiv_id":"2606.18524","last_updated":"2026-06-16T22:39:13Z","snapshot_observed_at":"2026-07-06T23:53:59.519305Z","submitted_at":"2026-06-16T22:39:13Z","title":"On the Residual Scaling of Looped Transformers: Stability and Transferability","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-27T00:53:51.358774Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2606.18524"},"observation_digest":"sha256:9fb795208835f51bda2807847b584913b6d39217d914c2e5e2196eb317fc7acd","observation_id":"c22b9625-6e5c-4e86-a5a5-87199693c6e7","resolution":{"observed_at":"2026-07-03T21:08:58.376035Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":"2410.05192","doi":"10.48550/arxiv.2410.05192","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding warmup-stable-decay learning rates: A river valley loss landscape perspective","venue":"arXiv (Cornell University)","work_id":"a4ab5b82-d4c2-4e3c-9f77-9c9de5253875","year":2024},"citing_paper":{"arxiv_id":"2606.18524","last_updated":"2026-06-16T22:39:13Z","snapshot_observed_at":"2026-07-06T23:53:59.519305Z","submitted_at":"2026-06-16T22:39:13Z","title":"On the Residual Scaling of Looped Transformers: Stability and Transferability","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-27T00:53:51.358774Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2606.18524"},"observation_digest":"sha256:a83a0b9bfad5d24f48731af70cac067f99e95b29f318ea04088b1956acfbea38","observation_id":"ff2bec40-5660-49f5-ad32-c9ab4dd77f72","resolution":{"observed_at":"2026-06-27T01:00:20.147453Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":"2410.05192","doi":"10.48550/arxiv.2410.05192","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding warmup-stable-decay learning rates: A river valley loss landscape perspective","venue":"arXiv (Cornell University)","work_id":"a4ab5b82-d4c2-4e3c-9f77-9c9de5253875","year":2024},"citing_paper":{"arxiv_id":"2606.21514","last_updated":"2026-06-19T15:10:20Z","snapshot_observed_at":"2026-07-06T23:56:31.454439Z","submitted_at":"2026-06-19T15:10:20Z","title":"Towards Understanding the Power and Limits of the Muon Optimizer: A River-Valley Perspective","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-26T14:22:54.988008Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2606.21514"},"observation_digest":"sha256:c2b58c45628fcbb35937a69d6fbbb7a5e2b7df79a90e11e784022f87710c9e44","observation_id":"1b7e697e-7f19-4b9d-a4ff-87dafb0c8516","resolution":{"observed_at":"2026-07-04T06:29:38.143264Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":"2410.05192","doi":"10.48550/arxiv.2410.05192","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding warmup-stable-decay learning rates: A river valley loss landscape perspective","venue":"arXiv (Cornell University)","work_id":"a4ab5b82-d4c2-4e3c-9f77-9c9de5253875","year":2024},"citing_paper":{"arxiv_id":"2606.25086","last_updated":"2026-06-30T16:49:44Z","snapshot_observed_at":"2026-08-07T14:30:24.863446Z","submitted_at":"2026-06-23T18:47:40Z","title":"Training for the Model You Return: Improving Optimization for Iterate-Averaged Language Models","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-26T00:09:54.353782Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2606.25086"},"observation_digest":"sha256:b61108ec824d9e96f3fb02fbcc01b47cdc6af394f89c9812d9738c3807544c56","observation_id":"b4d9028b-c0ea-4c28-b0ce-6b4a96fa1e93","resolution":{"observed_at":"2026-07-04T16:49:58.434221Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":"2410.05192","doi":"10.48550/arxiv.2410.05192","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding warmup-stable-decay learning rates: A river valley loss landscape perspective","venue":"arXiv (Cornell University)","work_id":"a4ab5b82-d4c2-4e3c-9f77-9c9de5253875","year":2024},"citing_paper":{"arxiv_id":"2606.25086","last_updated":"2026-06-30T16:49:44Z","snapshot_observed_at":"2026-08-07T14:30:24.863446Z","submitted_at":"2026-06-23T18:47:40Z","title":"Training for the Model You Return: Improving Optimization for Iterate-Averaged Language Models","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-07-01T06:36:46.924135Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2606.25086"},"observation_digest":"sha256:f0df8a87750eee42358f05bf069624655b3e04f71e3d8158bc99f95cabc1f65e","observation_id":"0d99bf80-4423-4c4e-915d-bf09ba765d4f","resolution":{"observed_at":"2026-07-01T09:25:41.300839Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-07-14T08:04:06.432613Z","title":"arXiv preprint arXiv:2410.05192 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.10959","last_updated":"2026-07-12T23:24:42Z","snapshot_observed_at":"2026-08-03T13:16:14.672928Z","submitted_at":"2026-07-12T23:24:42Z","title":"WSqD: A Horizon-Free Learning Rate Schedule for Large Model Training","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-07-14T08:04:06.432613Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2607.10959"},"observation_digest":"sha256:0382c95c5159204550f4f14b3a2c4869c5707611305d67aedb8f1bd960612e25","observation_id":"17f0a578-cff1-475f-b1ad-cad8eb94c624","resolution":{"observed_at":"2026-07-14T08:04:06.432613Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-01T12:16:39.111653Z","title":"2024, arXiv e-prints, arXiv:2410.05192","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.19640","last_updated":"2026-07-22T00:21:36Z","snapshot_observed_at":"2026-08-04T21:04:08.236868Z","submitted_at":"2026-07-22T00:21:36Z","title":"AGNFormer I: Reconstruction of AGN spectra using a probabilistic transformer model","version":1},"reference_index":101,"source":"arxiv_source","source_observed_at":"2026-08-01T12:16:39.111653Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2607.19640"},"observation_digest":"sha256:6842b2ba77342641bef5592c894eddb7166e29728bca16ecfa06bea109aac53c","observation_id":"f10ebdaa-b7d1-455d-80fb-0380eedbae66","resolution":{"observed_at":"2026-08-01T12:16:39.111653Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-01T12:16:45.544158Z","title":"arXiv e-prints , keywords =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.19640","last_updated":"2026-07-22T00:21:36Z","snapshot_observed_at":"2026-08-04T21:04:08.236868Z","submitted_at":"2026-07-22T00:21:36Z","title":"AGNFormer I: Reconstruction of AGN spectra using a probabilistic transformer model","version":1},"reference_index":153,"source":"arxiv_source","source_observed_at":"2026-08-01T12:16:45.544158Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2607.19640"},"observation_digest":"sha256:4eeb46d66de073e9f77cd9f5655e01bf0326486a301f9dd18a158ec4e84acdb9","observation_id":"651abc06-ade7-4366-9fc1-a69dde5ebb0e","resolution":{"observed_at":"2026-08-01T12:16:45.544158Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05192","snapshot_observed_at":"2026-08-01T08:50:45.185948Z","title":"arXiv preprint arXiv:2410.05192 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21005","last_updated":"2026-07-23T07:39:03Z","snapshot_observed_at":"2026-08-06T16:38:33.932762Z","submitted_at":"2026-07-23T07:39:03Z","title":"Weight-norm Criticality: A Mechanism for Loss Spikes Induced by the Normalization and Weight Decay","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-01T08:50:45.185948Z"},"links":{"cited_paper":"/paper/2410.05192","citing_paper":"/paper/2607.21005"},"observation_digest":"sha256:fde05bf48a3a9addc5284fcbdb332304c8a5c288353e28d3d7d0a0f42048ef98","observation_id":"6eb4aa11-6139-429c-b41c-b22a835280b6","resolution":{"observed_at":"2026-08-01T08:50:45.185948Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2410.05192/citation-record","integrity":"/paper/2410.05192/integrity","json":"/paper/2410.05192/citation-record.json","paper":"/paper/2410.05192"},"outbound":[],"paper":{"arxiv_id":"2410.05192","last_updated":"2024-12-02T21:54:54Z","latest_version":3,"primary_category":"cs.LG","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T16:49:39Z","title":"Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 37 inbound Pith citation observations for arXiv:2410.05192."}