{"as_of":"2026-08-10T07:20:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:fad3d9f9365aff08153e43c7be6823b13fa9ea5f32896c9c230697d2d34d36bb","coverage":[{"denominator":78,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":78,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T17:37:40.815997Z","state":"measured"},{"denominator":80,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":80,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-13T01:36:23.845366Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-13T01:36:23.932030Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"cited_work":{"arxiv_id":"2502.01694","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01694","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","venue":null,"work_id":"1f44a77e-59bd-478a-8c1b-709c1e247f29","year":2025},"citing_paper":{"arxiv_id":"2502.17419","last_updated":"2025-06-25T02:24:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-24T18:50:52Z","title":"From System 1 to System 2: A Survey of Reasoning Large Language Models","version":6},"reference_index":300,"source":"pdf_text","source_observed_at":"2026-05-13T01:36:23.845366Z"},"links":{"cited_paper":"/paper/2502.01694","citing_paper":"/paper/2502.17419"},"observation_digest":"sha256:4c118b561d80e2193f265da68edffafd0512f81cb4eee7c256850de62bdb6b0e","observation_id":"8c284df7-dd23-469d-8630-b8d80b83a028","resolution":{"observed_at":"2026-05-13T01:36:23.937668Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"cited_work":{"arxiv_id":"2502.01694","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01694","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","venue":null,"work_id":"1f44a77e-59bd-478a-8c1b-709c1e247f29","year":2025},"citing_paper":{"arxiv_id":"2605.10237","last_updated":"2026-05-11T09:11:20Z","snapshot_observed_at":"2026-07-06T23:22:13.774652Z","submitted_at":"2026-05-11T09:11:20Z","title":"The Benefits of Temporal Correlations: SGD Learns k-Juntas from Random Walks Efficiently","version":1},"reference_index":130,"source":"arxiv_source","source_observed_at":"2026-05-12T05:27:11.761971Z"},"links":{"cited_paper":"/paper/2502.01694","citing_paper":"/paper/2605.10237"},"observation_digest":"sha256:496bfcf494a419681a3a7c5f7d6d28a4767d03f975f4c6af79f00a866bda4882","observation_id":"2cd0d59c-5a84-4693-8361-82fbf94ab6ec","resolution":{"observed_at":"2026-05-12T05:31:24.330367Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2502.01694/citation-record","integrity":"/paper/2502.01694/integrity","json":"/paper/2502.01694/citation-record.json","paper":"/paper/2502.01694"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.533805Z","title":"How far can transformers reason? T he locality barrier and inductive scratchpad","venue":null,"work_id":"b1c4cdf5-e484-4d09-b562-466d882ffc8e","year":2024},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.569761Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:60702bd535cee837703d0b8f87390347a49d8b6dc88966689fda846b4ddcac9e","observation_id":"50c00ae1-3b43-4ffa-b4e7-ba111481a8b4","resolution":{"observed_at":"2026-08-09T17:37:41.536637Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.08073","last_updated":"2022-12-15T06:19:23Z","snapshot_observed_at":"2026-08-02T04:53:58.766070Z","submitted_at":"2022-12-15T06:19:23Z","title":"Constitutional AI: Harmlessness from AI Feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.08073","snapshot_observed_at":"2026-08-09T17:37:40.574169Z","title":"Constitutional AI : harmlessness from AI feedback","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.574169Z"},"links":{"cited_paper":"/paper/2212.08073","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:bdc64221ea498487f0a4b0b70f2f6558cbf9aa8290fa42542b26129a3f0d29f2","observation_id":"e438d207-4bb6-4b9f-896e-5d1448966b19","resolution":{"observed_at":"2026-08-09T17:37:40.574169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.525585Z","title":"Beltr\\' a n and C","venue":null,"work_id":"e560c728-de40-4f5f-84ef-f634b1275250","year":2011},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.578030Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:11c9089fa6320a6fdf054a560a0ae0fc267212d647f61f1484316bac41affd38","observation_id":"24a5fb29-86a9-4465-b1a8-8593739dab93","resolution":{"observed_at":"2026-08-09T17:37:41.528255Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.517284Z","title":"Graph of thoughts: solving elaborate problems with large language models","venue":null,"work_id":"294b154b-db3c-46dc-bf02-5738942f3483","year":2024},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.581505Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:796ed8c387850245014576a494f926d0a697c077b04aea00fb34d290a92fa1a8","observation_id":"ac51a1ab-8cde-4ef2-b566-066b90401c98","resolution":{"observed_at":"2026-08-09T17:37:41.520267Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.507853Z","title":"Multi-scale metastable dynamics and the asymptotic stationary distribution of perturbed M arkov chains","venue":null,"work_id":"fe461f90-8e70-4f3b-aa52-addaa6ca3bea","year":2016},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.585137Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:eb21c8eebf7a86964cadd6beed897e7680132016e746b848a79fab45b15c3867","observation_id":"0cb9b975-6509-42de-b57d-622616770bd3","resolution":{"observed_at":"2026-08-09T17:37:41.511374Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:40.588486Z","title":"Understanding in-context learning in transformers and LLM s by learning to learn discrete functions","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.588486Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:87f3aa26d39688dcaf2d435a7d1b1d7f2bc35ec94e775fdc088ed417901de245","observation_id":"2b994676-eff7-4a6f-8c49-7003c4e6a366","resolution":{"observed_at":"2026-08-09T17:37:40.588486Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.492347Z","title":"Metastable states, quasi-stationary distributions and soft measures","venue":null,"work_id":"585d66db-4d35-4402-b182-ac23c2cec4a5","year":2016},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.592115Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:cd20e23b617949dcade933e9bef267ebfd1a260a6376d5a227a8036a3686109e","observation_id":"6c81d042-285a-44ee-b136-05f8e29d1f98","resolution":{"observed_at":"2026-08-09T17:37:41.495805Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.482492Z","title":"Metastability and low lying spectra in reversible M arkov chains","venue":null,"work_id":"b98da2a9-9969-4738-89dd-d22729457190","year":2002},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.595430Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:775eb7b13e72794e3d44f1d9515aea8105231b591ac41822ef59e63c08dc2491","observation_id":"89100101-25d3-4c9c-86e4-ec4e3829ef17","resolution":{"observed_at":"2026-08-09T17:37:41.486093Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.472484Z","title":"On using extended statistical queries to avoid membership queries","venue":null,"work_id":"41220808-c5e3-4e95-a077-0e6e0c4217bf","year":2001},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.598732Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:ecf471bfd9be162d8468bc681d9f13bf0dbb660d7e01f521b859c3733c2eea06","observation_id":"9dd31d78-eb4b-4ea7-b163-a0b46229129f","resolution":{"observed_at":"2026-08-09T17:37:41.476089Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.463029Z","title":null,"venue":null,"work_id":"62ae1d70-d04c-45e3-9f8a-be6b1ffad84f","year":2018},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.602193Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:2a2fbe1dfe470f300023524043091f8ac58342aaa4d3ba901467af07e42b1273","observation_id":"54e4cb9c-3218-473e-a04b-8ab501f75d64","resolution":{"observed_at":"2026-08-09T17:37:41.466392Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:40.605429Z","title":"Exploration by random network distillation","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.605429Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:c7893d00c83f3ee97a8885b3865c3e24b33453fd6344fc4953b055fec58cac18","observation_id":"19df1bd3-babe-439a-aa57-3fb4c06146de","resolution":{"observed_at":"2026-08-09T17:37:40.605429Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.447370Z","title":"Tighter bounds on the expressivity of transformer encoders","venue":null,"work_id":"12272266-bd4a-44c7-91ec-94d5f4f952f3","year":2023},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.608516Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:ad4b8dd2ec9c5065d28e7c205b3f02bada5e94032a7d6b58ee093598a5181062","observation_id":"4aeef1d3-1ded-45f7-8343-e31d037a09e1","resolution":{"observed_at":"2026-08-09T17:37:41.450779Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.437680Z","title":"Metastability for general dynamics with rare transitions: escape time and critical configurations","venue":null,"work_id":"30838e44-acc0-4342-8755-e171f5dfe82f","year":2014},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.611157Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:5210187a690e153dfd32a4b1c9b2a632f49b3011b92fa33af308d2aa2cebb00d","observation_id":"f23df22b-6ad2-4664-b46d-00697d52e482","resolution":{"observed_at":"2026-08-09T17:37:41.441151Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-09T17:37:40.613806Z","title":"The L lama 3 herd of models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.613806Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:45c817139e87fd1782398c1ef789e23e2ae4424c18dbb2522766a2a341f70e6d","observation_id":"ec138d7e-1fef-4345-8cf3-21f1e2e22715","resolution":{"observed_at":"2026-08-09T17:37:40.613806Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.427695Z","title":"Edelman, eran malach, and Surbhi Goel","venue":null,"work_id":"185f0f15-8f18-4dfd-bb03-74235fa3846f","year":2024},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.616528Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:7e213050340cd3ef644757356382ba3ceaade73175177abb0a093b5557ef7675","observation_id":"edff7053-f4c8-4965-865a-2b817574dfaf","resolution":{"observed_at":"2026-08-09T17:37:41.431000Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.418177Z","title":null,"venue":null,"work_id":"2c8669e8-de4d-4fed-a447-4664524b3ed6","year":2018},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.619204Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:7085f183f7be8c093c863c4e9086d08bb209babe5eda7e5e32d3eefdfa15264c","observation_id":"3e56dd1e-f808-4a42-ba87-d0eac4056faf","resolution":{"observed_at":"2026-08-09T17:37:41.421609Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.408358Z","title":"A general characterization of the statistical query complexity","venue":null,"work_id":"7abbf5ff-d929-4d3d-8767-606659c9c0c0","year":2017},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.621676Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:9be7c7353b645e9f6a2bce021d47c5f53f8d9a74b428e98226d12ec87bfedf31","observation_id":"e2e99adf-8c49-403b-a03e-80512daef6c8","resolution":{"observed_at":"2026-08-09T17:37:41.411935Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.398002Z","title":"Towards revealing the mystery behind chain of thought: a theoretical perspective","venue":null,"work_id":"e86c3664-874e-42e3-9647-b916ab24a5ee","year":2023},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.624383Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:afd06f09d931d0bd9c6c6af523d27e12e69a237c87a1e8b5e9fda3f98a6cdc0a","observation_id":"89d70ac3-eb8a-4a3a-b481-e2b7952f8457","resolution":{"observed_at":"2026-08-09T17:37:41.401771Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17179","last_updated":"2024-02-09T00:13:46Z","snapshot_observed_at":"2026-07-06T16:25:25.534843Z","submitted_at":"2023-09-29T12:20:19Z","title":"Alphazero-like Tree-Search can Guide Large Language Model Decoding and Training","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.17179","snapshot_observed_at":"2026-08-09T17:37:40.626962Z","title":"Alphazero-like tree-search can guide large language model decoding and training","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.626962Z"},"links":{"cited_paper":"/paper/2309.17179","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:901cf3913a5cdd7269753a94b3c79d37c040f7c0a0f112f0a2a2f8626da908cd","observation_id":"469e47ad-84b8-460e-a008-a747f6fe8abb","resolution":{"observed_at":"2026-08-09T17:37:40.626962Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.388804Z","title":"Fernandez, F","venue":null,"work_id":"16ddba43-ed39-4ca7-a035-766d6bbb80d4","year":2016},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.629858Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:d80ed2c201effe8ac882aa3f610d0d44685d5617c9734a30dafde1e7a1f72ace","observation_id":"071849db-828d-4540-8e35-26bf1c123f34","resolution":{"observed_at":"2026-08-09T17:37:41.391558Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.380033Z","title":"Asymptotically exponential hitting times and metastability: A pathwise approach without reversibility","venue":null,"work_id":"e9c0d57a-ec5c-4bd3-97be-3af9ac8bdb04","year":2014},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.632315Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:a8918cdd76044cbba6db7237d4d1f75e60ca12fffdb32455436ecece45e26698","observation_id":"7e8eda49-89cb-4d9e-8040-264ffc98f8e1","resolution":{"observed_at":"2026-08-09T17:37:41.382935Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.371540Z","title":"An SVD approach to identifying metastable states of Markov chains","venue":null,"work_id":"1c7230f1-1500-4a65-a004-4607b9a303b3","year":2008},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.635584Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:0da878b74b0a97c9cc30421fcf53007ebdde6f06d5f9532e3d0546be5ebacf31","observation_id":"d0e991fa-b97f-4c3b-bcd4-5eeb67c1279d","resolution":{"observed_at":"2026-08-09T17:37:41.374387Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03683","last_updated":"2024-04-01T06:50:52Z","snapshot_observed_at":"2026-08-10T01:12:38.149323Z","submitted_at":"2024-04-01T06:50:52Z","title":"Stream of Search (SoS): Learning to Search in Language","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.03683","snapshot_observed_at":"2026-08-09T17:37:40.638911Z","title":"Stream of search ( SoS ): learning to search in language","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.638911Z"},"links":{"cited_paper":"/paper/2404.03683","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:5c86fafe0755d8ed660a19c5ddd1a8713c31b857e5e41298ed32fdbf3f38b747","observation_id":"48240f96-30af-43c8-8308-ccd706082de6","resolution":{"observed_at":"2026-08-09T17:37:40.638911Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-09T17:37:40.642300Z","title":"DeepSeek-R1: incentivizing reasoning capability in LLMs via reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.642300Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:10c2e65ad6c892d5fd25da298a93f86d4fe9d06e81a619aa3e3eca00cc40cf94","observation_id":"b92c176c-6045-4369-901f-0c9a3b86708b","resolution":{"observed_at":"2026-08-09T17:37:40.642300Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.15556","last_updated":"2022-03-29T13:38:03Z","snapshot_observed_at":"2026-08-09T19:52:33.533277Z","submitted_at":"2022-03-29T13:38:03Z","title":"Training Compute-Optimal Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.15556","snapshot_observed_at":"2026-08-09T17:37:40.645993Z","title":"Training compute-optimal large language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.645993Z"},"links":{"cited_paper":"/paper/2203.15556","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:68892d116dcb92d586d32ade4d720ad9fb04fcd0ebc58ba81d9b94d26d910083","observation_id":"dd70058d-78f5-45b0-a30b-b72a19a6cdbe","resolution":{"observed_at":"2026-08-09T17:37:40.645993Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.02301","last_updated":"2023-07-05T16:59:31Z","snapshot_observed_at":"2026-07-06T15:22:55.122322Z","submitted_at":"2023-05-03T17:50:56Z","title":"Distilling Step-by-Step! Outperforming Larger Language Models with Less Training Data and Smaller Model Sizes","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.02301","snapshot_observed_at":"2026-08-09T17:37:40.649626Z","title":"Distilling step-by-step! O utperforming larger language models with less training data and smaller model sizes","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.649626Z"},"links":{"cited_paper":"/paper/2305.02301","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:1a6a46da2f4f45b09e8e1d16c4a257d9ca252c9cf0c9eabf318fb283396f646e","observation_id":"e5b84f39-108a-4f7f-99bd-ec0a02abb40a","resolution":{"observed_at":"2026-08-09T17:37:40.649626Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14511","last_updated":"2024-08-28T14:13:41Z","snapshot_observed_at":"2026-08-09T19:52:27.492309Z","submitted_at":"2024-08-25T04:07:18Z","title":"Unveiling the Statistical Foundations of Chain-of-Thought Prompting Methods","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.14511","snapshot_observed_at":"2026-08-09T17:37:40.653083Z","title":"Unveiling the statistical foundations of chain-of-thought prompting methods","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.653083Z"},"links":{"cited_paper":"/paper/2408.14511","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:e0b5f2aef0bcb1662a3024e156320943bc27a52c68dc3dfb1fe4c25d5406970b","observation_id":"5a07c5ea-c178-4f41-9fb7-c849e8acde89","resolution":{"observed_at":"2026-08-09T17:37:40.653083Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.362819Z","title":"From self-attention to M arkov models: unveiling the dynamics of generative transformers","venue":null,"work_id":"968a1442-bf5c-4be8-98f1-52a91832fbb6","year":2024},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.656125Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:4e68d8ea6fb1f127cdb93d831164db164b578dda049eac1deb55460584f6ad11","observation_id":"b12e8f29-6e43-4430-ad5b-4bfd4ec8ae77","resolution":{"observed_at":"2026-08-09T17:37:41.365773Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.352880Z","title":"A robust spectral method for finding lumpings and meta-stable states of non-reversible M arkov chains","venue":null,"work_id":"a5ea6f19-a277-45e8-b072-50eea06fc3c3","year":2010},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.659523Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:4715319d116723c903f651d81a3e4052adb92dd1f948739d91e2deffb800f894","observation_id":"7afa93a0-6c1a-49e0-b280-44042fba37ea","resolution":{"observed_at":"2026-08-09T17:37:41.356263Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16720","last_updated":"2026-04-30T02:46:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-21T18:04:31Z","title":"OpenAI o1 System Card","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.16720","snapshot_observed_at":"2026-08-09T17:37:40.662426Z","title":"OpenAI o1 system card","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.662426Z"},"links":{"cited_paper":"/paper/2412.16720","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:b7522fe2314d1036dbbde84cc18a28e6bcd9eb93ab1c501cd9b2aec2b413cce5","observation_id":"37bf7a01-0b54-4bc9-b596-a91a54369a6f","resolution":{"observed_at":"2026-08-09T17:37:40.662426Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1803.07300","last_updated":"2019-06-08T13:57:05Z","snapshot_observed_at":"2026-08-09T02:59:09.657776Z","submitted_at":"2018-03-20T08:47:27Z","title":"Risk and parameter convergence of logistic regression","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1803.07300","snapshot_observed_at":"2026-08-09T17:37:40.665817Z","title":"Risk and parameter convergence of logistic regression","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.665817Z"},"links":{"cited_paper":"/paper/1803.07300","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:906a3e6e8d8460b71e80f25cae58844ece8d178c846a3d74cc38a55dd44a050a","observation_id":"e7498104-799f-4444-986d-865692d73cf5","resolution":{"observed_at":"2026-08-09T17:37:40.665817Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.03113","last_updated":"2021-04-15T10:03:37Z","snapshot_observed_at":"2026-07-06T10:57:14.668681Z","submitted_at":"2021-04-07T13:34:25Z","title":"Scaling Scaling Laws with Board Games","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.03113","snapshot_observed_at":"2026-08-09T17:37:40.669105Z","title":"Scaling scaling laws with board games","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.669105Z"},"links":{"cited_paper":"/paper/2104.03113","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:a574adf72627f6310116e02abec7e0f8774a9efd457c6521008a5159dde31307","observation_id":"7fe2369e-b6da-4d2c-995e-96ac8a535b11","resolution":{"observed_at":"2026-08-09T17:37:40.669105Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:40.672379Z","title":"Thinking, fast and slow","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.672379Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:230784cc95323ff75b76cd4bbcebacaa742271bf563db1d461063bd5830c9d93","observation_id":"2417e27c-3988-47b8-b0f1-fe2d0c01df37","resolution":{"observed_at":"2026-08-09T17:37:40.672379Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2001.08361","last_updated":"2020-01-23T03:59:20Z","snapshot_observed_at":"2026-07-06T08:52:12.656082Z","submitted_at":"2020-01-23T03:59:20Z","title":"Scaling Laws for Neural Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2001.08361","snapshot_observed_at":"2026-08-09T17:37:40.675692Z","title":"Scaling laws for neural language models","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.675692Z"},"links":{"cited_paper":"/paper/2001.08361","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:eb46a0a0b0146f4dbad36ebedf47270a737cd0cb3c5dec85b3e91694160380d0","observation_id":"b6918358-61f6-47a2-a91f-5ff3a45743a7","resolution":{"observed_at":"2026-08-09T17:37:40.675692Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.337496Z","title":"Efficient noise-tolerant learning from statistical queries","venue":null,"work_id":"3710704e-02be-4490-8c50-5c09a9f05f3a","year":1998},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.678892Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:bbec5f781715f110be80a89ed967bada54416f15eabd692d160db0d4d4a39dca","observation_id":"c345ce1e-e0f3-40ec-9b08-2b61dd640948","resolution":{"observed_at":"2026-08-09T17:37:41.341073Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.08633","last_updated":"2025-03-11T14:26:41Z","snapshot_observed_at":"2026-08-09T13:53:56.927549Z","submitted_at":"2024-10-11T08:55:17Z","title":"Transformers Provably Solve Parity Efficiently with Chain of Thought","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.08633","snapshot_observed_at":"2026-08-09T17:37:40.682061Z","title":"Transformers provably solve parity efficiently with chain of thought","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.682061Z"},"links":{"cited_paper":"/paper/2410.08633","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:dfadaa3b055578cfbfb278f5274382f641cbf9ee50252775ced274de1379989f","observation_id":"5faf387d-a41d-4915-a41a-42faf968e9ea","resolution":{"observed_at":"2026-08-09T17:37:40.682061Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12599","last_updated":"2025-06-03T02:14:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T02:48:14Z","title":"Kimi k1.5: Scaling Reinforcement Learning with LLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12599","snapshot_observed_at":"2026-08-09T17:37:40.685445Z","title":"Kimi k1.5: scaling reinforcement learning with LLM s","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.685445Z"},"links":{"cited_paper":"/paper/2501.12599","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:5d266bb47d5953c7e62115b749b0da77d8273d5635d906496bb59818ea431238","observation_id":"30225a18-51ae-4175-88cd-af4693d0674d","resolution":{"observed_at":"2026-08-09T17:37:40.685445Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12917","last_updated":"2024-10-04T17:28:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-19T17:16:21Z","title":"Training Language Models to Self-Correct via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12917","snapshot_observed_at":"2026-08-09T17:37:40.688957Z","title":"Training language models to self-correct via reinforcement learning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.688957Z"},"links":{"cited_paper":"/paper/2409.12917","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:f9170008d486ed3f688fbe735b92086d8026f1ca281af0bd7e8b3dbbfddcae41","observation_id":"8b264acb-2949-47cf-9403-c39bc63f3a80","resolution":{"observed_at":"2026-08-09T17:37:40.688957Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.328254Z","title":null,"venue":null,"work_id":"b4c87401-e8b0-48fe-abf4-16f63addda94","year":2012},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.692341Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:1b895ca1315bace94dd2aa45884bd7bfa09570cf650153a141fa5d1673684ab8","observation_id":"08acdfda-e33b-48b1-847e-7e9b986c704b","resolution":{"observed_at":"2026-08-09T17:37:41.331402Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1807.04144","last_updated":"2018-07-11T14:04:21Z","snapshot_observed_at":"2026-07-06T06:49:37.294670Z","submitted_at":"2018-07-11T14:04:21Z","title":"Metastable Markov chains","version":1},"cited_work":{"arxiv_id":"1807.04144","doi":null,"metadata_source":"pith","pith_arxiv_id":"1807.04144","snapshot_observed_at":"2026-08-09T17:37:41.016844Z","title":"Metastable Markov chains","venue":"math.PR","work_id":"3d803878-70c7-407a-8eb5-62c0cbacdc53","year":2018},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.695286Z"},"links":{"cited_paper":"/paper/1807.04144","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:3c1ede77f53c78f6b3f3247dcfd40bcad9d92603af21f50388c4e215777aff86","observation_id":"03e0c432-004e-4585-8ef0-1c9ef6347810","resolution":{"observed_at":"2026-08-09T17:37:41.020448Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.318710Z","title":"Metastability of finite state M arkov chains: A recursive procedure to identify slow variables for model reduction","venue":null,"work_id":"3b276347-ae1d-49ae-8f81-9921a7fd727e","year":2015},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.698504Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:05d8e2c494f481fabc296308a29fe794bc4f52fbef6780eb23a61fd72aa16e0e","observation_id":"12268893-11f3-41f1-8e37-ac1dcf0009e5","resolution":{"observed_at":"2026-08-09T17:37:41.322027Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.309158Z","title":"Markov Chains and Mixing Times","venue":null,"work_id":"fecca78c-bed8-47da-9820-bb4318e02d49","year":2009},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.701182Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:b7a575c6ef9ecfaa251dc7950a2b50bdc63b4841a5a64a2c52a2573523fe6bdf","observation_id":"77b3410a-cd18-4ff9-8885-affb135f9885","resolution":{"observed_at":"2026-08-09T17:37:41.312481Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.299527Z","title":"How do nonlinear transformers acquire generalization-guaranteed CoT ability? In High-dimensional Learning Dynamics 2024: The Emergence of Structure and Reasoning, 2024 a","venue":null,"work_id":"8cf82cbc-93d4-41d7-b5d0-ad97c0f32240","year":2024},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.703758Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:bd2b3082c61fac1d42cd5e9febe4bd982b404975fcfd74b0918238c469795546","observation_id":"2f7d99d4-2f58-4def-a562-a386a6d5ff2c","resolution":{"observed_at":"2026-08-09T17:37:41.303024Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.289881Z","title":"Dissecting chain-of-thought: compositionality through in-context filtering and learning","venue":null,"work_id":"589ef30f-096b-4521-bbbd-78fc16ee2a64","year":2023},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.706269Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:eb94ca2b057a037a287d7a69ff2b0b6407faf3c1e4322a39e10e77f81195cfb0","observation_id":"5d094478-9b26-4cf9-99d2-26cd5fd48885","resolution":{"observed_at":"2026-08-09T17:37:41.293444Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.12875","last_updated":"2024-09-21T06:48:45Z","snapshot_observed_at":"2026-08-07T02:31:13.044154Z","submitted_at":"2024-02-20T10:11:03Z","title":"Chain of Thought Empowers Transformers to Solve Inherently Serial Problems","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.12875","snapshot_observed_at":"2026-08-09T17:37:40.708818Z","title":"Chain of thought empowers transformers to solve inherently serial problems","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.708818Z"},"links":{"cited_paper":"/paper/2402.12875","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:cb3182ba2a0cf143bddea3f7df20b39e680fe4671ea61cba885a3f0b59155718","observation_id":"b353fa84-b2d0-4ea2-ab11-22513e1d936e","resolution":{"observed_at":"2026-08-09T17:37:40.708818Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.20050","last_updated":"2023-05-31T17:24:00Z","snapshot_observed_at":"2026-08-05T13:11:04.104454Z","submitted_at":"2023-05-31T17:24:00Z","title":"Let's Verify Step by Step","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.20050","snapshot_observed_at":"2026-08-09T17:37:40.711533Z","title":"Let's verify step by step","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.711533Z"},"links":{"cited_paper":"/paper/2305.20050","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:2267201c801e227bd13bb6d232df22f211e85b8735a57f719e81d7647ef673d7","observation_id":"24774810-46a9-4268-9075-14a390b6444b","resolution":{"observed_at":"2026-08-09T17:37:40.711533Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.280141Z","title":"Markov chain decomposition for convergence rate analysis","venue":null,"work_id":"ef7f3ea6-fe03-4ccf-b38f-ace3ec32689b","year":2001},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.714354Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:85c25a262421789cadf2fd1f009b4733ab412a27e94b8ca2b313ff2c3f407eda","observation_id":"9961a831-396c-4a27-a69d-1b8d65c0a58e","resolution":{"observed_at":"2026-08-09T17:37:41.283333Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04161","last_updated":"2025-07-21T14:23:40Z","snapshot_observed_at":"2026-08-07T02:03:46.144376Z","submitted_at":"2024-02-06T17:18:59Z","title":"Attention with Markov: A Framework for Principled Analysis of Transformers via Markov Chains","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.04161","snapshot_observed_at":"2026-08-09T17:37:40.717508Z","title":"Attention with M arkov: A framework for principled analysis of transformers via M arkov chains","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.717508Z"},"links":{"cited_paper":"/paper/2402.04161","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:02188f4c71f01973785a7b1861d0551ad9ba02f7e8d88107af61f48da302f2b1","observation_id":"5c5eceea-8ea7-477f-87b3-9750e7750afd","resolution":{"observed_at":"2026-08-09T17:37:40.717508Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.07923","last_updated":"2024-04-11T18:03:53Z","snapshot_observed_at":"2026-08-02T23:51:22.170619Z","submitted_at":"2023-10-11T22:35:18Z","title":"The Expressive Power of Transformers with Chain of Thought","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.07923","snapshot_observed_at":"2026-08-09T17:37:40.720858Z","title":"The expresssive power of transformers with chain of thought","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.720858Z"},"links":{"cited_paper":"/paper/2310.07923","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:0144fa54cf42f3a8daebf61335e540c4c162f43b14ec4e210c9aa677df32eb50","observation_id":"ac800233-a78a-4198-a1c7-f973b4176e20","resolution":{"observed_at":"2026-08-09T17:37:40.720858Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.270914Z","title":null,"venue":null,"work_id":"fba9118d-f407-4656-a7ba-df76cedb0c88","year":1989},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.724628Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:8f4bbfe6b04948480184cb9abeee8e1d2d926f370ff56b2f7b7447ae59e955cb","observation_id":"65e29515-9988-4385-95a3-0bdd90c367e6","resolution":{"observed_at":"2026-08-09T17:37:41.274007Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.261357Z","title":"The dondition of a finite M arkov chain and perturbation bounds for the limiting probabilities","venue":null,"work_id":"846300dc-34b9-4b9b-975c-827b549a7ce9","year":1980},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.727588Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:e9e38bf79f77023ca2cf7ecb91a733dc0b7f24db20f454723923e4cebef0359b","observation_id":"995399be-ef35-42b2-8eba-547f5edfb5ee","resolution":{"observed_at":"2026-08-09T17:37:41.264868Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14735","last_updated":"2024-08-13T15:45:37Z","snapshot_observed_at":"2026-07-06T17:34:07.737296Z","submitted_at":"2024-02-22T17:47:03Z","title":"How Transformers Learn Causal Structure with Gradient Descent","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14735","snapshot_observed_at":"2026-08-09T17:37:40.730834Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.730834Z"},"links":{"cited_paper":"/paper/2402.14735","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:f80e7898acc9fa30232ff4a224d688573a3015c788d4fe05c28d25a24a8e9371","observation_id":"b4b8a0cc-5aff-4b01-91e0-42c87d69b3f1","resolution":{"observed_at":"2026-08-09T17:37:40.730834Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.00114","last_updated":"2021-11-30T21:32:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-11-30T21:32:46Z","title":"Show Your Work: Scratchpads for Intermediate Computation with Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.00114","snapshot_observed_at":"2026-08-09T17:37:40.734065Z","title":"Show your work: scratchpads for intermediate computation with language models","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.734065Z"},"links":{"cited_paper":"/paper/2112.00114","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:c7f768bac55489425f0bb6f0570f198fbac266ca4ebd43443a5cbd3a8c371762","observation_id":"308c9202-df0e-4a94-b4cc-e293a707d617","resolution":{"observed_at":"2026-08-09T17:37:40.734065Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.251964Z","title":"Spinning up: proximal policy optimization ( PPO ), 2018","venue":null,"work_id":"41a25e26-c644-4fec-b29f-f4922750a149","year":2018},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.737533Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:44cc665af84d4fbdcb43860ee44257dcbca8566d136ef2d22643b1fd8bdfe1c3","observation_id":"b7ab6864-ae78-42e4-bb54-d28681e259c2","resolution":{"observed_at":"2026-08-09T17:37:41.255318Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.243111Z","title":"Concentration inequalities for Markov chains by Marton couplings and spectral methods","venue":null,"work_id":"541f5266-dcd2-4701-aec9-bcad39b1fe99","year":2015},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.740784Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:dfd89d40c44e2a4fee98cf4af53573e96cb7b85c6813f38f6dbb26a1fe1a7dce","observation_id":"f4159ad9-0374-4bf2-a78f-3e6e8c63a2d1","resolution":{"observed_at":"2026-08-09T17:37:41.245878Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:40.743850Z","title":"Improving language understanding by generative pre-training","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.743850Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:ae755162d149db220bc0118288982122ae7b2fd4dff47661b11c396a5fc7c677","observation_id":"744c92bc-a181-4673-ac27-48d0f55c058c","resolution":{"observed_at":"2026-08-09T17:37:40.743850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.18512","last_updated":"2024-05-28T18:31:14Z","snapshot_observed_at":"2026-07-06T18:21:35.476139Z","submitted_at":"2024-05-28T18:31:14Z","title":"Understanding Transformer Reasoning Capabilities via Graph Algorithms","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.18512","snapshot_observed_at":"2026-08-09T17:37:40.747108Z","title":"Understanding transformer reasoning capabilities via graph algorithms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.747108Z"},"links":{"cited_paper":"/paper/2405.18512","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:e35b5ccaf32e9ddd91859f081504873c3b82970e488bb984ceb9efd8dd755fa3","observation_id":"2669e4e7-1f0e-4ab0-8527-94308893de45","resolution":{"observed_at":"2026-08-09T17:37:40.747108Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.228750Z","title":"Transformers, parallel computation, and logarithmic depth","venue":null,"work_id":"e7703408-f6fa-4daa-be01-de69f49b5847","year":2024},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.750617Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:b87cdf1f88647418ffd8fd41a75489d5d2041c300850558ca9afeea9580479dc","observation_id":"9a4f38dc-8c81-4cbb-a9c8-51502a1928b1","resolution":{"observed_at":"2026-08-09T17:37:41.231676Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-09T17:37:40.753852Z","title":"Proximal policy optimization algorithms","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.753852Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:1fe08918a0c876a49f1579ec082fc560e94fa4da8ff6f90fca40c8f687febb13","observation_id":"968d5237-196b-42c2-9c15-56e5adc7d758","resolution":{"observed_at":"2026-08-09T17:37:40.753852Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.220647Z","title":"Failures of gradient-based deep learning","venue":null,"work_id":"11993698-8595-488a-a728-e1945f4ff11d","year":2017},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.757299Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:7d6bb98d603c7afc4e64ae86742961c461621568b717569f7e58c88a307871a1","observation_id":"61f1fcab-3a01-473e-a46b-d9ff406aabc6","resolution":{"observed_at":"2026-08-09T17:37:41.223371Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.211785Z","title":"Distribution-specific hardness of learning neural networks","venue":null,"work_id":"c7c2794a-ee75-4f65-bddd-5bbbda7ab380","year":2018},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.760427Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:116ae37af86faaec1f658f02dac2d292424dfedded7028fb9a2a271a88c8a53b","observation_id":"91772f9d-6a93-4736-8c97-60796dc9d278","resolution":{"observed_at":"2026-08-09T17:37:41.214982Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.00193","last_updated":"2023-05-18T04:44:51Z","snapshot_observed_at":"2026-08-09T20:40:15.403863Z","submitted_at":"2022-12-01T00:39:56Z","title":"Distilling Reasoning Capabilities into Smaller Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.00193","snapshot_observed_at":"2026-08-09T17:37:40.763610Z","title":"Distilling reasoning capabilities into smaller language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.763610Z"},"links":{"cited_paper":"/paper/2212.00193","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:e7f62c2b3eaf3198aaeb0be7bb9fbd1a8dcffb7ea83bffc6afb484355f61a270","observation_id":"c7785893-e39f-48a0-bc0d-a4e16d8f8e84","resolution":{"observed_at":"2026-08-09T17:37:40.763610Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.202398Z","title":"A general reinforcement learning algorithm that masters chess, shogi, and G o through self-play","venue":null,"work_id":"27336352-74dc-4233-9afb-61f14960ba19","year":2018},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.766915Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:854ce5b55b7db9552e61d8d7213df283243959c30edf11b563b2522b01393495","observation_id":"296fcd87-0aae-4d82-af24-2e38d886710c","resolution":{"observed_at":"2026-08-09T17:37:41.205822Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03314","last_updated":"2024-08-06T17:35:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:35:05Z","title":"Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03314","snapshot_observed_at":"2026-08-09T17:37:40.770120Z","title":"Scaling LLM test-time compute optimally can be more effective than scaling model parameters","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.770120Z"},"links":{"cited_paper":"/paper/2408.03314","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:28ef780b569f3519989a6fc256f8999bb50c6ff4f086be9b60efb0b8be62a073","observation_id":"5bb19291-011c-4e56-bafb-bb65cecde4fe","resolution":{"observed_at":"2026-08-09T17:37:40.770120Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.192592Z","title":"On an SVD-based algorithm for identifying meta-stable states of Markov chains","venue":null,"work_id":"5d50af69-546d-4b35-9b2c-7b92f803ccc7","year":2011},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.773565Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:ca94993c2f38039753c8a13329d87c8d8d16e23bb719d5e6c41de51825bfbafb","observation_id":"000de0fa-d41b-4a9d-8255-83c582ff224b","resolution":{"observed_at":"2026-08-09T17:37:41.196240Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:40.777217Z","title":"Solving olympiad geometry without human demonstrations","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.777217Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:1e0d8996984043f1079f992f67f2682eb6e127739faf64b41740b2721dbde47f","observation_id":"8166e16f-0343-40fe-806b-e150555df2f0","resolution":{"observed_at":"2026-08-09T17:37:40.777217Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2211.14275","last_updated":"2022-11-25T18:19:44Z","snapshot_observed_at":"2026-08-01T02:16:43.109337Z","submitted_at":"2022-11-25T18:19:44Z","title":"Solving math word problems with process- and outcome-based feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.14275","snapshot_observed_at":"2026-08-09T17:37:40.780527Z","title":"Solving math word problems with process-and outcome-based feedback","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.780527Z"},"links":{"cited_paper":"/paper/2211.14275","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:be7c3d9acc572c551f6cd17f771af9e1f408d725969f5d9979d0d15755652778","observation_id":"ad2d2566-6750-4f48-9ff7-74600fb4dbbc","resolution":{"observed_at":"2026-08-09T17:37:40.780527Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:40.784127Z","title":"Chain-of-thought prompting elicits reasoning in large language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.784127Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:87f3b2bf9bad89c16328d84a79e4cb402dca1bea5a64f88c17f509f0507b4412","observation_id":"f90ef72d-ac50-4ee3-8860-8b9b19b37220","resolution":{"observed_at":"2026-08-09T17:37:40.784127Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05459","last_updated":"2025-03-05T13:57:56Z","snapshot_observed_at":"2026-07-06T19:29:18.770662Z","submitted_at":"2024-10-07T19:45:09Z","title":"From Sparse Dependence to Sparse Attention: Unveiling How Chain-of-Thought Enhances Transformer Sample Efficiency","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05459","snapshot_observed_at":"2026-08-09T17:37:40.787506Z","title":"From sparse dependence to sparse attention: unveiling how chain-of-thought enhances transformer sample efficiency","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.787506Z"},"links":{"cited_paper":"/paper/2410.05459","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:3a6cd01fee551f70dfb5ed2aca204bada7cc1f3a1759ebfa15a35a66f46cddd5","observation_id":"7dda2a9b-e9a1-410b-b2db-4868a9765058","resolution":{"observed_at":"2026-08-09T17:37:40.787506Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.172256Z","title":"An algorithm for computing stochastically stable distributions with applications to mltiagent learning in repeated games","venue":null,"work_id":"3db93e63-7d42-480f-8096-3f597acc4730","year":2005},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.790869Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:2b34da4b14871a54e844d24632d0bbf22995c5d3a8eb60eaffe930c3544c2ef2","observation_id":"0c50f4c7-5063-4dec-9504-5d105b643444","resolution":{"observed_at":"2026-08-09T17:37:41.175635Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1902.01224","last_updated":"2022-08-16T09:18:49Z","snapshot_observed_at":"2026-07-06T07:30:57.901088Z","submitted_at":"2019-02-01T08:13:58Z","title":"Estimating the Mixing Time of Ergodic Markov Chains","version":4},"cited_work":{"arxiv_id":"1902.01224","doi":null,"metadata_source":"pith","pith_arxiv_id":"1902.01224","snapshot_observed_at":"2026-08-09T17:37:40.891536Z","title":"Estimating the Mixing Time of Ergodic Markov Chains","venue":"math.ST","work_id":"d509b0d8-e468-4e86-bd88-628b57ac94ff","year":2019},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.793651Z"},"links":{"cited_paper":"/paper/1902.01224","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:251b1cfe39bb711fff41246a15160a77f4e572cfb17e91c20e8ff43f82819ad2","observation_id":"3a93b245-11d4-4632-909d-d86fde763e0a","resolution":{"observed_at":"2026-08-09T17:37:40.897167Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00724","last_updated":"2025-03-03T07:53:32Z","snapshot_observed_at":"2026-08-09T23:53:42.648697Z","submitted_at":"2024-08-01T17:16:04Z","title":"Inference Scaling Laws: An Empirical Analysis of Compute-Optimal Inference for Problem-Solving with Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00724","snapshot_observed_at":"2026-08-09T17:37:40.796752Z","title":"Inference scaling laws: An empirical analysis of compute-optimal inference for problem-solving with language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.796752Z"},"links":{"cited_paper":"/paper/2408.00724","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:cb1b3e3c389389ce99bc6c55e1ad567c3caa86b1c24d940ae13e81b09068cdea","observation_id":"f916fd35-a510-4a5d-a42b-d1c159d687b9","resolution":{"observed_at":"2026-08-09T17:37:40.796752Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.04682","last_updated":"2025-01-08T18:42:48Z","snapshot_observed_at":"2026-07-06T20:18:21.419068Z","submitted_at":"2025-01-08T18:42:48Z","title":"Towards System 2 Reasoning in LLMs: Learning How to Think With Meta Chain-of-Thought","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.04682","snapshot_observed_at":"2026-08-09T17:37:40.799741Z","title":"Towards System 2 reasoning in LLMs: learning how to think with meta chain-of-thought","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.799741Z"},"links":{"cited_paper":"/paper/2501.04682","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:0b430b6a27d33c65821281e28e027e85d0f327ef38c89ed7cb34283f068c3692","observation_id":"1d7d5871-f5a4-48c6-919a-71dabc5f150f","resolution":{"observed_at":"2026-08-09T17:37:40.799741Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-09T04:48:29.237275Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-09T17:37:40.802953Z","title":"Lillicrap, Kenji Kawaguchi, and Michael Shieh","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.802953Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:5e8721f1a6e85010026c77ae5e18379d825c55625f0b31e4e3f0798c11f5fafe","observation_id":"77bd2389-0d2d-41bf-8f95-2dc1e2df1484","resolution":{"observed_at":"2026-08-09T17:37:40.802953Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1905.13211","last_updated":"2020-02-15T06:56:25Z","snapshot_observed_at":"2026-07-06T07:56:50.324784Z","submitted_at":"2019-05-30T17:53:30Z","title":"What Can Neural Networks Reason About?","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1905.13211","snapshot_observed_at":"2026-08-09T17:37:40.806200Z","title":"What can neural networks reason about? arXiv preprint arXiv:1905.13211, 2019","venue":null,"work_id":null,"year":1905},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.806200Z"},"links":{"cited_paper":"/paper/1905.13211","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:049a42a5703a228b551dd02540c91fb26c2eae4dfa5a97db69cb83dce98321ea","observation_id":"8e536755-9a40-4a33-93ff-19d5b89a3da3","resolution":{"observed_at":"2026-08-09T17:37:40.806200Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:40.809759Z","title":"Tree of thoughts: deliberate problem solving with large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.809759Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:7b83201c8ff5d671dde408b49bd1d4f779e2bb7925b6b7837187a25b60b19a45","observation_id":"65f680de-6147-4b82-9b70-1f33cafdff9b","resolution":{"observed_at":"2026-08-09T17:37:40.809759Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02724","last_updated":"2025-02-02T15:57:01Z","snapshot_observed_at":"2026-08-06T13:43:29.144376Z","submitted_at":"2024-10-03T17:45:31Z","title":"Large Language Models as Markov Chains","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02724","snapshot_observed_at":"2026-08-09T17:37:40.812795Z","title":"Large language models as M arkov chains","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.812795Z"},"links":{"cited_paper":"/paper/2410.02724","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:3d2911832865e23fdb9b2f437c8405d52071c28d1735d3a92733e90cd617288e","observation_id":"4eb8bfad-a55f-4ce3-9f4f-03d7de9d50fc","resolution":{"observed_at":"2026-08-09T17:37:40.812795Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T17:37:41.157270Z","title":"Star: bootstrapping reasoning with reasoning","venue":null,"work_id":"7bff2fe7-9125-451b-b0c7-aa0075ad0ae9","year":2022},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.815997Z"},"links":{"citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:800f2ab2644406e5ac2ef9be64979b4f3a0f69b9f7a7121d6a7cc30eacbad9e9","observation_id":"c58d727f-2cd0-43b7-af55-33038c0db0f3","resolution":{"observed_at":"2026-08-09T17:37:41.160931Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","latest_version":2,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-09T20:41:13.212163Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation"},"reference_resolution":{"displayed":78,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":43,"verified_exact":2,"verified_fuzzy":33},"total_outbound_references":78},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 78 of 78 outbound references and 2 inbound Pith citation observations for arXiv:2502.01694."}