{"as_of":"2026-08-13T23:56:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:38471084e68c31478671af1c40c248f4ed3b4bdd7b5ce14bfa3b65bd927610cd","coverage":[{"denominator":62,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":62,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T23:15:28.793471Z","state":"measured"},{"denominator":80,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":80,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":18,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":18,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T14:54:29.270448Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-01T20:56:13.476979Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02674","snapshot_observed_at":"2026-08-09T14:54:29.270448Z","title":"Mind the gap: Examining the self-improvement capabilities of large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01612","last_updated":"2025-02-13T05:32:54Z","snapshot_observed_at":"2026-08-13T19:07:17.024757Z","submitted_at":"2025-02-03T18:45:22Z","title":"Self-Improving Transformers Overcome Easy-to-Hard and Length Generalization Challenges","version":2},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-09T14:54:29.270448Z"},"links":{"cited_paper":"/paper/2412.02674","citing_paper":"/paper/2502.01612"},"observation_digest":"sha256:f53a968f4ee05ef06db287deb5d0ea7a5485c6f84b3c900b56ba1df31777c322","observation_id":"574ba37f-3705-454b-b24d-4ca3b9a0819a","resolution":{"observed_at":"2026-08-09T14:54:29.270448Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02674","snapshot_observed_at":"2026-08-07T10:28:22.812009Z","title":"Mind the gap: Examining the self-improvement capabilities of large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05109","last_updated":"2025-06-05T14:53:35Z","snapshot_observed_at":"2026-08-13T01:01:54.935509Z","submitted_at":"2025-06-05T14:53:35Z","title":"Truly Self-Improving Agents Require Intrinsic Metacognitive Learning","version":1},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-07T10:28:22.812009Z"},"links":{"cited_paper":"/paper/2412.02674","citing_paper":"/paper/2506.05109"},"observation_digest":"sha256:95d3f813832d0e0142b5511aef297a81650cb3a5eded797ba4c1f65eee91c92c","observation_id":"e61b5bbf-b799-4ccf-ab58-29688b857d9b","resolution":{"observed_at":"2026-08-07T10:28:22.812009Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02674","snapshot_observed_at":"2026-08-07T10:35:38.336495Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05295","last_updated":"2025-06-12T16:25:06Z","snapshot_observed_at":"2026-08-11T19:02:17.096406Z","submitted_at":"2025-06-05T17:48:19Z","title":"Sample Complexity and Representation Ability of Test-time Scaling Paradigms","version":2},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-07T10:35:38.336495Z"},"links":{"cited_paper":"/paper/2412.02674","citing_paper":"/paper/2506.05295"},"observation_digest":"sha256:0c7092ed81e2516661561534d1c22f7fddf5ec299da1858cfe5911fef3f0e581","observation_id":"1c0905cb-9aa6-41d5-ac77-19d69cfcfcb6","resolution":{"observed_at":"2026-08-07T10:35:38.336495Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02674","snapshot_observed_at":"2026-08-07T05:03:28.400160Z","title":"Mind the gap: Examining the self-improvement capabilities of large language models.arXiv preprint arXiv:2412.02674, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.09026","last_updated":"2025-06-13T17:44:03Z","snapshot_observed_at":"2026-08-12T20:27:13.801957Z","submitted_at":"2025-06-10T17:52:42Z","title":"e3: Learning to Explore Enables Extrapolation of Test-Time Compute for LLMs","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T05:03:28.400160Z"},"links":{"cited_paper":"/paper/2412.02674","citing_paper":"/paper/2506.09026"},"observation_digest":"sha256:a80c50fc7e9bfe7b4108b97171c18369e1b16a33e619698238c21b3b26115b63","observation_id":"b760d7ee-a2d3-4a3e-a999-b4747dad6391","resolution":{"observed_at":"2026-08-07T05:03:28.400160Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02674","snapshot_observed_at":"2026-08-07T04:39:36.497124Z","title":"Mind the gap: Examining the self-improvement capabilities of large language models.arXiv preprint arXiv:2412.02674, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10406","last_updated":"2025-06-12T06:59:35Z","snapshot_observed_at":"2026-08-08T09:07:12.052492Z","submitted_at":"2025-06-12T06:59:35Z","title":"PAG: Multi-Turn Reinforced LLM Self-Correction with Policy as Generative Verifier","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T04:39:36.497124Z"},"links":{"cited_paper":"/paper/2412.02674","citing_paper":"/paper/2506.10406"},"observation_digest":"sha256:53fd89204985775c4788eea523634407b3009e500113221214a9b9cc7fd4651a","observation_id":"ecfaabd5-011e-4fb5-8f41-d6591eece8e5","resolution":{"observed_at":"2026-08-07T04:39:36.497124Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02674","snapshot_observed_at":"2026-08-07T01:03:21.917517Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12152","last_updated":"2025-06-13T18:13:58Z","snapshot_observed_at":"2026-08-10T03:56:47.932564Z","submitted_at":"2025-06-13T18:13:58Z","title":"Because we have LLMs, we Can and Should Pursue Agentic Interpretability","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T01:03:21.917517Z"},"links":{"cited_paper":"/paper/2412.02674","citing_paper":"/paper/2506.12152"},"observation_digest":"sha256:5db337666a8bfde0a49ca629713de6f10e2929073b6cfe122c2d3531d073b6ff","observation_id":"eb7f58e0-4da1-417f-8b70-2b500455401a","resolution":{"observed_at":"2026-08-07T01:03:21.917517Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02674","snapshot_observed_at":"2026-08-07T00:31:36.251369Z","title":"Qwen Team","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.13639","last_updated":"2025-06-16T16:04:43Z","snapshot_observed_at":"2026-08-07T00:26:19.217552Z","submitted_at":"2025-06-16T16:04:43Z","title":"An Empirical Study of LLM-as-a-Judge: How Design Choices Impact Evaluation Reliability","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T00:31:36.251369Z"},"links":{"cited_paper":"/paper/2412.02674","citing_paper":"/paper/2506.13639"},"observation_digest":"sha256:d7c649b317e32709c1bf1d57bf7520c0d3401e556b80e61d78a219f32097f18f","observation_id":"e2a5c216-4747-45e8-91ec-72ca80aed3cd","resolution":{"observed_at":"2026-08-07T00:31:36.251369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02674","snapshot_observed_at":"2026-08-05T13:52:52.207779Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.00271","last_updated":"2026-06-17T20:45:44Z","snapshot_observed_at":"2026-08-08T09:36:37.528608Z","submitted_at":"2025-08-29T22:56:32Z","title":"Learn from What We HAVE: History-Aware VErifier that Reasons about Past Interactions Online","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-05T13:52:52.207779Z"},"links":{"cited_paper":"/paper/2412.02674","citing_paper":"/paper/2509.00271"},"observation_digest":"sha256:1a8020bfeb503db385b90a1024835bb6c7b599b1c1483e28736f618fb0a225a3","observation_id":"d1eb8d0e-a29f-4d6d-8e30-6db51c75e167","resolution":{"observed_at":"2026-08-05T13:52:52.207779Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02674","snapshot_observed_at":"2026-08-04T22:59:14.556101Z","title":"The importance of online data: Understanding preference fine-tuning via coverage.Advances in Neural Information Processing Systems, 37:12243–12270, 2024a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.06941","last_updated":"2025-09-08T17:52:56Z","snapshot_observed_at":"2026-08-07T12:11:53.268632Z","submitted_at":"2025-09-08T17:52:56Z","title":"Outcome-based Exploration for LLM Reasoning","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-04T22:59:14.556101Z"},"links":{"cited_paper":"/paper/2412.02674","citing_paper":"/paper/2509.06941"},"observation_digest":"sha256:02ab89c83cb4f4ec415f1e73aa4aaf23d31096cedbd9e8a38af01a1a0a5ffae6","observation_id":"73d87d62-3600-436e-8f4c-b412c9fef2d6","resolution":{"observed_at":"2026-08-04T22:59:14.556101Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02674","snapshot_observed_at":"2026-08-03T02:43:36.764821Z","title":"Mind the gap: Examining the self-improvement capabilities of large language models.arXiv preprint arXiv:2412.02674, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.10014","last_updated":"2026-05-31T19:13:22Z","snapshot_observed_at":"2026-08-06T02:34:35.293978Z","submitted_at":"2026-02-10T17:36:41Z","title":"A Task-Centric Theory for Iterative Self-Improvement with Easy-to-Hard Curricula","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-03T02:43:36.764821Z"},"links":{"cited_paper":"/paper/2412.02674","citing_paper":"/paper/2602.10014"},"observation_digest":"sha256:6b80b39b8939768ac592994e95254df74eeea0019cf39a9550e19953dc07f4c0","observation_id":"88fa9f2e-85a3-466f-8bbc-a2d61ed641f9","resolution":{"observed_at":"2026-08-03T02:43:36.764821Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"cited_work":{"arxiv_id":"2412.02674","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.02674","snapshot_observed_at":"2026-07-01T20:56:13.476979Z","title":"Mind the gap: Examining the self-improvement capabilities of large language models","venue":null,"work_id":"13db825e-1742-4503-aff5-6755fdc61624","year":2025},"citing_paper":{"arxiv_id":"2604.16335","last_updated":"2026-03-13T02:23:49Z","snapshot_observed_at":"2026-08-04T04:51:19.798134Z","submitted_at":"2026-03-13T02:23:49Z","title":"Beyond Verifiable Rewards: Rubric-Based GRM for Reinforced Fine-Tuning SWE Agents","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-15T12:22:13.551709Z"},"links":{"cited_paper":"/paper/2412.02674","citing_paper":"/paper/2604.16335"},"observation_digest":"sha256:31fe2e98dd9873ed772b34bcd6e80652943760ee8b25897516add349437e3718","observation_id":"3c17bca8-58cc-4883-b5af-fff3617f1794","resolution":{"observed_at":"2026-05-15T12:25:35.849997Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"cited_work":{"arxiv_id":"2412.02674","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.02674","snapshot_observed_at":"2026-07-01T20:56:13.476979Z","title":"Mind the gap: Examining the self-improvement capabilities of large language models","venue":null,"work_id":"13db825e-1742-4503-aff5-6755fdc61624","year":2025},"citing_paper":{"arxiv_id":"2604.16804","last_updated":"2026-05-06T17:41:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-18T03:24:54Z","title":"AutoOR: Scalably Post-training LLMs to Autoformalize Operations Research Problems","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-10T07:02:02.992871Z"},"links":{"cited_paper":"/paper/2412.02674","citing_paper":"/paper/2604.16804"},"observation_digest":"sha256:55e97bdc25b6367eaf514e6aee8508b573a41efa75ff876ea6c0c566167860b4","observation_id":"d4970fb6-29dc-4158-9ff4-388f7f142f42","resolution":{"observed_at":"2026-05-10T07:11:53.591039Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"cited_work":{"arxiv_id":"2412.02674","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.02674","snapshot_observed_at":"2026-07-01T20:56:13.476979Z","title":"Mind the gap: Examining the self-improvement capabilities of large language models","venue":null,"work_id":"13db825e-1742-4503-aff5-6755fdc61624","year":2025},"citing_paper":{"arxiv_id":"2604.17089","last_updated":"2026-04-18T17:58:37Z","snapshot_observed_at":"2026-07-06T23:04:14.688737Z","submitted_at":"2026-04-18T17:58:37Z","title":"Tree of Concepts: Interpretable Continual Learners in Non-Stationary Clinical Domains","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T06:44:44.353815Z"},"links":{"cited_paper":"/paper/2412.02674","citing_paper":"/paper/2604.17089"},"observation_digest":"sha256:15528c0bdd394f4676adc9a9787d54b4704de46b58cb9f5d248c16f5fc7c51c6","observation_id":"fc296c37-263e-4b37-918d-b29e828bd96c","resolution":{"observed_at":"2026-05-10T06:46:37.231503Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"cited_work":{"arxiv_id":"2412.02674","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.02674","snapshot_observed_at":"2026-07-01T20:56:13.476979Z","title":"Mind the gap: Examining the self-improvement capabilities of large language models","venue":null,"work_id":"13db825e-1742-4503-aff5-6755fdc61624","year":2025},"citing_paper":{"arxiv_id":"2605.09995","last_updated":"2026-05-11T05:11:04Z","snapshot_observed_at":"2026-08-10T20:47:41.204946Z","submitted_at":"2026-05-11T05:11:04Z","title":"Annotations Mitigate Post-Training Mode Collapse","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-05-12T03:58:11.179607Z"},"links":{"cited_paper":"/paper/2412.02674","citing_paper":"/paper/2605.09995"},"observation_digest":"sha256:59bef885ba7ccb6fbcd08f9f9a057e3d70b8bc4f33dde55d8908db4f920ae8de","observation_id":"caec8726-3d6a-4f00-a7cb-0274bf57bdcc","resolution":{"observed_at":"2026-05-12T06:46:51.283999Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"cited_work":{"arxiv_id":"2412.02674","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.02674","snapshot_observed_at":"2026-07-01T20:56:13.476979Z","title":"Mind the gap: Examining the self-improvement capabilities of large language models","venue":null,"work_id":"13db825e-1742-4503-aff5-6755fdc61624","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-08-08T11:14:54.052517Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2412.02674","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:a4831483e7c8e8c183b753e2bc91e43cecfb124406b4d845ed65febe1936ec37","observation_id":"f5591126-1c6c-4e21-b113-78511853ae51","resolution":{"observed_at":"2026-06-28T17:22:24.928244Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"cited_work":{"arxiv_id":"2412.02674","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.02674","snapshot_observed_at":"2026-07-01T20:56:13.476979Z","title":"Mind the gap: Examining the self-improvement capabilities of large language models","venue":null,"work_id":"13db825e-1742-4503-aff5-6755fdc61624","year":2025},"citing_paper":{"arxiv_id":"2606.01249","last_updated":"2026-06-17T04:44:10Z","snapshot_observed_at":"2026-08-12T03:28:30.781632Z","submitted_at":"2026-05-31T14:04:51Z","title":"Trust Region On-Policy Distillation","version":3},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-06-28T17:38:50.313305Z"},"links":{"cited_paper":"/paper/2412.02674","citing_paper":"/paper/2606.01249"},"observation_digest":"sha256:7a84a247a56340419958fec0ce894358beac3003454e89008bf5eb0bd035b689","observation_id":"95b3bb05-48ac-4514-b2bf-632c27e7fbed","resolution":{"observed_at":"2026-07-01T20:56:13.478449Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"cited_work":{"arxiv_id":"2412.02674","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.02674","snapshot_observed_at":"2026-07-01T20:56:13.476979Z","title":"Mind the gap: Examining the self-improvement capabilities of large language models","venue":null,"work_id":"13db825e-1742-4503-aff5-6755fdc61624","year":2025},"citing_paper":{"arxiv_id":"2606.31511","last_updated":"2026-06-30T11:26:14Z","snapshot_observed_at":"2026-08-12T16:49:08.020465Z","submitted_at":"2026-06-30T11:26:14Z","title":"Falsification, Not Exposure: An Internally Preregistered Placebo-Controlled Decomposition of Self-Repair Feedback in Frozen Small Code Models","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-07-01T04:44:56.520156Z"},"links":{"cited_paper":"/paper/2412.02674","citing_paper":"/paper/2606.31511"},"observation_digest":"sha256:560c207d97e7d7674a2f787591adb056af73dccb7566678141e2635b69193120","observation_id":"6d689bb3-6ea7-4f52-9251-8247dbb3f52d","resolution":{"observed_at":"2026-07-01T11:05:42.232371Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02674","snapshot_observed_at":"2026-08-06T00:42:58.750607Z","title":"Mind the gap: Examining the self-improvement capabilities of large language models.arXiv preprint arXiv:2412.02674, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.01000","last_updated":"2026-08-02T05:00:44Z","snapshot_observed_at":"2026-08-11T20:46:23.964034Z","submitted_at":"2026-08-02T05:00:44Z","title":"Judging Is Not Enumerating: Silent Omissions in LLM-Authored Acceptable Sets","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T00:42:58.750607Z"},"links":{"cited_paper":"/paper/2412.02674","citing_paper":"/paper/2608.01000"},"observation_digest":"sha256:d2a8b7f02e3be2ab68f9e1886260baac57ef4f074060078817cbfe10b9193268","observation_id":"4a151dce-1fbe-4ea6-af68-2b1378e91128","resolution":{"observed_at":"2026-08-06T00:42:58.750607Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2412.02674/citation-record","integrity":"/paper/2412.02674/integrity","json":"/paper/2412.02674/citation-record.json","paper":"/paper/2412.02674"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-11T23:15:28.483490Z","title":"Gpt-4 technical report.arXiv preprint arXiv:2303.08774,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.483490Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:6121af27a228fbba93175a371efd3a14e740e05f33053d022a9a092a7570e53e","observation_id":"a582408f-2d25-4c20-9cc5-d3096012ef66","resolution":{"observed_at":"2026-08-11T23:15:28.483490Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.11791","last_updated":"2024-08-21T17:24:15Z","snapshot_observed_at":"2026-08-12T22:59:48.671593Z","submitted_at":"2024-08-21T17:24:15Z","title":"Critique-out-Loud Reward Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.11791","snapshot_observed_at":"2026-08-11T23:15:28.498365Z","title":"Critique- out-loud reward models.arXiv preprint arXiv:2408.11791,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.498365Z"},"links":{"cited_paper":"/paper/2408.11791","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:998989565ed68f5382551425b600dd598da8c8c1c23b4c9df9cca790034a2c1d","observation_id":"1006b770-934f-41e5-9ed3-7cad476dce90","resolution":{"observed_at":"2026-08-11T23:15:28.498365Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16609","last_updated":"2023-09-28T17:07:49Z","snapshot_observed_at":"2026-08-09T21:25:20.369782Z","submitted_at":"2023-09-28T17:07:49Z","title":"Qwen Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16609","snapshot_observed_at":"2026-08-11T23:15:28.503131Z","title":"Qwen technical report.arXiv preprint arXiv:2309.16609,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.503131Z"},"links":{"cited_paper":"/paper/2309.16609","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:9f0611e5ff29c31ef8421b8338a6010579164d6d872e1d389a9d19501109434e","observation_id":"3e86cb86-5877-4f2d-a9bc-0c27bb0838af","resolution":{"observed_at":"2026-08-11T23:15:28.503131Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.08073","last_updated":"2022-12-15T06:19:23Z","snapshot_observed_at":"2026-08-02T04:53:58.766070Z","submitted_at":"2022-12-15T06:19:23Z","title":"Constitutional AI: Harmlessness from AI Feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.08073","snapshot_observed_at":"2026-08-11T23:15:28.508973Z","title":"Constitutional ai: Harmlessness from ai feedback","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.508973Z"},"links":{"cited_paper":"/paper/2212.08073","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:44ce1382e71e7d5f73572f8350bc26190f0c5e15a212ba282dfb7c4fcc9e1d23","observation_id":"4e21ec20-274a-43bf-9623-e75a19af96ee","resolution":{"observed_at":"2026-08-11T23:15:28.508973Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.00429","last_updated":"2024-04-02T14:09:40Z","snapshot_observed_at":"2026-08-13T10:00:51.371111Z","submitted_at":"2023-09-30T16:41:04Z","title":"On the Stability of Iterative Retraining of Generative Models on their own Data","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.00429","snapshot_observed_at":"2026-08-11T23:15:28.525342Z","title":"On the stability of iterative retraining of generative models on their own data.arXiv preprint arXiv:2310.00429,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.525342Z"},"links":{"cited_paper":"/paper/2310.00429","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:cbf6d23d34f50b303b9f811f08ba467a29cc4adfb64f888f81abccdb3c6cee3a","observation_id":"8cecfc31-dcfe-4552-9b88-b57bf309ce46","resolution":{"observed_at":"2026-08-11T23:15:28.525342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16822","last_updated":"2024-06-17T07:07:30Z","snapshot_observed_at":"2026-08-13T05:15:23.224755Z","submitted_at":"2023-11-28T14:36:43Z","title":"Large Language Models Suffer From Their Own Output: An Analysis of the Self-Consuming Training Loop","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16822","snapshot_observed_at":"2026-08-11T23:15:28.530737Z","title":"Large language models suffer from their own output: An analysis of the self-consuming training loop.arXiv preprint arXiv:2311.16822,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.530737Z"},"links":{"cited_paper":"/paper/2311.16822","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:742b664e1add86caed9d90933d5ad6b174679afe65ed8e535d91a67db1584e76","observation_id":"d2c8edaf-45f1-49d0-9a5a-abaa4a59eb3e","resolution":{"observed_at":"2026-08-11T23:15:28.530737Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21787","last_updated":"2024-12-30T19:03:24Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:57:25Z","title":"Large Language Monkeys: Scaling Inference Compute with Repeated Sampling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21787","snapshot_observed_at":"2026-08-11T23:15:28.535758Z","title":"Large language monkeys: Scaling inference compute with repeated sampling.arXiv preprint arXiv:2407.21787,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.535758Z"},"links":{"cited_paper":"/paper/2407.21787","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:a4854a713ff0cd4650fce5738559afb651d152ee5c8bb2d412b175760c609a4c","observation_id":"7a626e7b-1941-48ca-aa31-42cb9b82f236","resolution":{"observed_at":"2026-08-11T23:15:28.535758Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.12712","last_updated":"2023-04-13T20:41:31Z","snapshot_observed_at":"2026-08-03T04:49:15.195814Z","submitted_at":"2023-03-22T16:51:28Z","title":"Sparks of Artificial General Intelligence: Early experiments with GPT-4","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.12712","snapshot_observed_at":"2026-08-11T23:15:28.539819Z","title":"Sparks of artificial general intelligence: Early experiments with gpt-4.arXiv preprint arXiv:2303.12712,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.539819Z"},"links":{"cited_paper":"/paper/2303.12712","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:677386e6956586d660958c9629a9cab50ef4f724f25cfa31793081929149b963","observation_id":"2eabb5c7-befa-4308-9e34-e7a3f7b09e84","resolution":{"observed_at":"2026-08-11T23:15:28.539819Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.11816","last_updated":"2023-11-13T18:51:42Z","snapshot_observed_at":"2026-08-13T11:13:06.992714Z","submitted_at":"2023-06-20T18:19:17Z","title":"Learning to Generate Better Than Your LLM","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.11816","snapshot_observed_at":"2026-08-11T23:15:28.551006Z","title":"Learning to generate better than your llm.arXiv preprint arXiv:2306.11816,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.551006Z"},"links":{"cited_paper":"/paper/2306.11816","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:5a40c5c4ea9220e1e89d9d227484d0fdd4e91bc53c17c244a57cd85f26a881e1","observation_id":"4b86df85-5d4a-4114-bf8b-29e86fa1edcd","resolution":{"observed_at":"2026-08-11T23:15:28.551006Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05128","last_updated":"2023-10-05T09:12:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-11T10:43:43Z","title":"Teaching Large Language Models to Self-Debug","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.05128","snapshot_observed_at":"2026-08-11T23:15:28.554478Z","title":"Teaching large language models to self-debug","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.554478Z"},"links":{"cited_paper":"/paper/2304.05128","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:8ed46e686422dec322ebae8edeaf4183344208cddd9d5780ab357d48b6009b55","observation_id":"d31c9d90-01d5-4efa-a92b-d52e383a60f7","resolution":{"observed_at":"2026-08-11T23:15:28.554478Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.01335","last_updated":"2024-06-14T21:17:17Z","snapshot_observed_at":"2026-08-13T19:59:28.723365Z","submitted_at":"2024-01-02T18:53:13Z","title":"Self-Play Fine-Tuning Converts Weak Language Models to Strong Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.01335","snapshot_observed_at":"2026-08-11T23:15:28.559268Z","title":"Self-play fine-tuning converts weak language models to strong language models.arXiv preprint arXiv:2401.01335,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.559268Z"},"links":{"cited_paper":"/paper/2401.01335","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:18f551d5d73c6a99f6fbd051e8daccdaef94420eb20c48e8bfc8a9e955dab35c","observation_id":"49dffea3-7aed-42a6-aa7f-9ed251e93a59","resolution":{"observed_at":"2026-08-11T23:15:28.559268Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.01937","last_updated":"2023-05-03T07:28:50Z","snapshot_observed_at":"2026-08-13T11:50:17.843572Z","submitted_at":"2023-05-03T07:28:50Z","title":"Can Large Language Models Be an Alternative to Human Evaluations?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.01937","snapshot_observed_at":"2026-08-11T23:15:28.563407Z","title":"Can large language models be an alternative to human evaluations? arXiv preprint arXiv:2305.01937,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.563407Z"},"links":{"cited_paper":"/paper/2305.01937","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:200b5c6fb9b95589cd7963b31ee3f2d9e113602021d31cd10e36d67909e6bd7c","observation_id":"e7f030f7-35dc-4570-97ee-63008b05c85a","resolution":{"observed_at":"2026-08-11T23:15:28.563407Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:15:28.567793Z","title":"Vicuna: An open-source chatbot impressing gpt-4 with 90%* chatgpt quality.See https://vicuna","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.567793Z"},"links":{"citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:a12aece25156627246b4a03cc692b137ce732c1cf91c8b89643b1ad315307d5d","observation_id":"97831683-ac83-4a52-96f8-61f4818844d7","resolution":{"observed_at":"2026-08-11T23:15:28.567793Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07712","last_updated":"2024-04-30T18:03:13Z","snapshot_observed_at":"2026-08-13T04:20:33.390256Z","submitted_at":"2024-02-12T15:26:01Z","title":"Model Collapse Demystified: The Case of Regression","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07712","snapshot_observed_at":"2026-08-11T23:15:28.579987Z","title":"Model collapse demystified: The case of regression.arXiv preprint arXiv:2402.07712,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.579987Z"},"links":{"cited_paper":"/paper/2402.07712","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:9c5d589a134d81600458335ed05e4886aea2f3a1edebefb9e45add37871732b2","observation_id":"9d8a2194-e9ad-46f9-9fa4-ba7564fe7149","resolution":{"observed_at":"2026-08-11T23:15:28.579987Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.07890","last_updated":"2025-04-21T16:43:00Z","snapshot_observed_at":"2026-08-12T23:24:46.969145Z","submitted_at":"2024-07-10T17:57:58Z","title":"Training on the Test Task Confounds Evaluation and Emergence","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.07890","snapshot_observed_at":"2026-08-11T23:15:28.585346Z","title":"Training on the test task confounds evaluation and emergence.arXiv preprint arXiv:2407.07890,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.585346Z"},"links":{"cited_paper":"/paper/2407.07890","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:a7c6754bd873c46d877efdfae9afd97a1741cc087d5bae66623b445b183dcf36","observation_id":"bb192b01-f8f8-47b4-831f-9bd212ebdeaf","resolution":{"observed_at":"2026-08-11T23:15:28.585346Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.07863","last_updated":"2024-11-12T11:18:43Z","snapshot_observed_at":"2026-08-13T14:44:02.569922Z","submitted_at":"2024-05-13T15:50:39Z","title":"RLHF Workflow: From Reward Modeling to Online RLHF","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.07863","snapshot_observed_at":"2026-08-11T23:15:28.589443Z","title":"Rlhf workflow: From reward modeling to online rlhf.arXiv preprint arXiv:2405.07863,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.589443Z"},"links":{"cited_paper":"/paper/2405.07863","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:b0f47df3abd04435185b98948c5d9aba775ec073c4ee2643359a1aaf02f352fb","observation_id":"28e71f36-6cbb-4fcd-998d-7ad8038de913","resolution":{"observed_at":"2026-08-11T23:15:28.589443Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-11T23:15:28.593584Z","title":"The llama 3 herd of models.arXiv preprint arXiv:2407.21783,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.593584Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:5f7365e3b3fb2a39c4a7f7364f6bbb459d285f9fad9b2af95d0db5a14c51497b","observation_id":"20974d57-4de3-4937-bda5-070ac236be73","resolution":{"observed_at":"2026-08-11T23:15:28.593584Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.04475","last_updated":"2025-03-10T09:27:03Z","snapshot_observed_at":"2026-07-06T17:56:23.317089Z","submitted_at":"2024-04-06T02:29:02Z","title":"Length-Controlled AlpacaEval: A Simple Way to Debias Automatic Evaluators","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.04475","snapshot_observed_at":"2026-08-11T23:15:28.598804Z","title":"Length-controlled alpacaeval: A simple way to debias automatic evaluators.arXiv preprint arXiv:2404.04475,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.598804Z"},"links":{"cited_paper":"/paper/2404.04475","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:42ba638a1a51bef202f4171af18e5f1ae9e6bdd7ca9c70f3ce21156a66d2ac83","observation_id":"b14dd521-d354-43a2-8a13-bd9516519e42","resolution":{"observed_at":"2026-08-11T23:15:28.598804Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:15:28.603090Z","title":"Matthias Gerstgrasser, Rylan Schaeffer, Apratim Dey, Rafael Rafailov, Henry Sleight, John Hughes, Tomasz Korbak, Rajashree Agrawal, Dhruv Pai, Andrey Gromov, et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.603090Z"},"links":{"citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:161bf9fb05612ca640973ba68b0498070d31054aa639cccfcbcd8abc571c0fa9","observation_id":"ed77794f-ce2e-484c-8aff-76c5130c7452","resolution":{"observed_at":"2026-08-11T23:15:28.603090Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07087","last_updated":"2024-06-10T14:22:45Z","snapshot_observed_at":"2026-08-13T04:21:54.240246Z","submitted_at":"2024-02-11T02:34:42Z","title":"Self-Correcting Self-Consuming Loops for Generative Model Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07087","snapshot_observed_at":"2026-08-11T23:15:28.607337Z","title":"Self-correcting self-consuming loops for generative model training.arXiv preprint arXiv:2402.07087,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.607337Z"},"links":{"cited_paper":"/paper/2402.07087","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:ab54a54aaf62864fff71e54ffd7f6691d63ab13091b236f187636003ade57e6d","observation_id":"95e0dace-5d23-469c-a1dd-ff4d14cc32ae","resolution":{"observed_at":"2026-08-11T23:15:28.607337Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.11644","last_updated":"2023-10-02T06:12:30Z","snapshot_observed_at":"2026-08-13T11:19:55.436754Z","submitted_at":"2023-06-20T16:14:25Z","title":"Textbooks Are All You Need","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.11644","snapshot_observed_at":"2026-08-11T23:15:28.611461Z","title":"Textbooks are all you need","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.611461Z"},"links":{"cited_paper":"/paper/2306.11644","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:0f6d444d65332bd3bc0abaf29aa07295c68768d9fa852647411ca72ac628b187","observation_id":"c7419b19-0da9-4ea2-9723-92b4175c42b8","resolution":{"observed_at":"2026-08-11T23:15:28.611461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04642","last_updated":"2024-03-07T16:36:29Z","snapshot_observed_at":"2026-08-13T00:59:58.573803Z","submitted_at":"2024-03-07T16:36:29Z","title":"Teaching Large Language Models to Reason with Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04642","snapshot_observed_at":"2026-08-11T23:15:28.615413Z","title":"Teaching large language models to reason with reinforcement learning.arXiv preprint arXiv:2403.04642,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.615413Z"},"links":{"cited_paper":"/paper/2403.04642","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:3830fee2d651bd2e88e4dbefaa5d7ee6d2e91c2a1689ea4b2297295c43836aad","observation_id":"71f6d78d-5621-4854-a1ab-cd6d43502736","resolution":{"observed_at":"2026-08-11T23:15:28.615413Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:15:28.637902Z","title":"Scaling laws for downstream task performance of large language models.arXiv preprint arXiv:2402.04177,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.637902Z"},"links":{"citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:d1d88ad9019eeb6bcf7756933b0e458373df6542a5531837c0e41ada67d9b83b","observation_id":"99c1ad99-4373-41cc-84e8-58185689d9f1","resolution":{"observed_at":"2026-08-11T23:15:28.637902Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.04298","last_updated":"2024-09-06T01:14:26Z","snapshot_observed_at":"2026-08-13T00:37:24.530514Z","submitted_at":"2024-04-04T20:27:37Z","title":"SELF-[IN]CORRECT: LLMs Struggle with Discriminating Self-Generated Responses","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.04298","snapshot_observed_at":"2026-08-11T23:15:28.641881Z","title":"Self-[in] correct: Llms struggle with refining self-generated responses.arXiv preprint arXiv:2404.04298,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.641881Z"},"links":{"cited_paper":"/paper/2404.04298","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:6e4d34f74fa871566dc57940138bac2bfcba3447be459661ccd82174665c017e","observation_id":"4ba54d39-b856-4fbb-985f-6dd250ead3cd","resolution":{"observed_at":"2026-08-11T23:15:28.641881Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2001.08361","last_updated":"2020-01-23T03:59:20Z","snapshot_observed_at":"2026-08-13T17:41:53.092611Z","submitted_at":"2020-01-23T03:59:20Z","title":"Scaling Laws for Neural Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2001.08361","snapshot_observed_at":"2026-08-11T23:15:28.649181Z","title":"Scaling laws for neural language models.arXiv preprint arXiv:2001.08361,","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.649181Z"},"links":{"cited_paper":"/paper/2001.08361","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:dcf6d751353ae221af34dbea6e2744f71612b92306eb6ac9f5dbbf35d4c8c18b","observation_id":"f6b417b9-0794-4870-8504-6ea296da7493","resolution":{"observed_at":"2026-08-11T23:15:28.649181Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.13064","last_updated":"2024-02-20T15:00:35Z","snapshot_observed_at":"2026-08-13T04:14:27.645630Z","submitted_at":"2024-02-20T15:00:35Z","title":"Synthetic Data (Almost) from Scratch: Generalized Instruction Tuning for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.13064","snapshot_observed_at":"2026-08-11T23:15:28.658554Z","title":"Synthetic data (almost) from scratch: Generalized instruction tuning for language models.arXiv preprint arXiv:2402.13064,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.658554Z"},"links":{"cited_paper":"/paper/2402.13064","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:c7f919776c43f2261bba05a38bb868681ec912d3658409fabff086a54e7441cf","observation_id":"75b51781-c29e-4c73-9188-0be6463c098a","resolution":{"observed_at":"2026-08-11T23:15:28.658554Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.05463","last_updated":"2023-09-11T14:01:45Z","snapshot_observed_at":"2026-08-02T22:47:03.212781Z","submitted_at":"2023-09-11T14:01:45Z","title":"Textbooks Are All You Need II: phi-1.5 technical report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.05463","snapshot_observed_at":"2026-08-11T23:15:28.662986Z","title":"Textbooks are all you need ii: phi-1.5 technical report.arXiv preprint arXiv:2309.05463,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.662986Z"},"links":{"cited_paper":"/paper/2309.05463","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:4638bc1cbd4b6bdbc7fa584e18ad831a0608bf2be9532c069de3bb4b3e8e7585","observation_id":"8b5a27ef-cb02-45dc-a395-dbd640ca7882","resolution":{"observed_at":"2026-08-11T23:15:28.662986Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.08072","last_updated":"2024-12-17T07:30:54Z","snapshot_observed_at":"2026-08-12T23:03:13.984289Z","submitted_at":"2024-08-15T10:44:38Z","title":"I-SHEEP: Self-Alignment of LLM from Scratch through an Iterative Self-Enhancement Paradigm","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.08072","snapshot_observed_at":"2026-08-11T23:15:28.667466Z","title":"I-sheep: Self-alignment of llm from scratch through an iterative self-enhancement paradigm","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.667466Z"},"links":{"cited_paper":"/paper/2408.08072","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:46cd1841da702b11ab739b74047906d484bb6738e4e7f393adafee318d261152","observation_id":"393ae529-41ec-4015-aede-d8de3469f8a7","resolution":{"observed_at":"2026-08-11T23:15:28.667466Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.20050","last_updated":"2023-05-31T17:24:00Z","snapshot_observed_at":"2026-08-11T17:22:43.545531Z","submitted_at":"2023-05-31T17:24:00Z","title":"Let's Verify Step by Step","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.20050","snapshot_observed_at":"2026-08-11T23:15:28.672176Z","title":"Let’s verify step by step.arXiv preprint arXiv:2305.20050,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.672176Z"},"links":{"cited_paper":"/paper/2305.20050","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:38c430e1b18912e0933bcb42f0e1c886a3c3c1bd595352c01aba92ddd88a51cd","observation_id":"ff6abce8-9a92-4643-ae6c-6ceb759e6950","resolution":{"observed_at":"2026-08-11T23:15:28.672176Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06657","last_updated":"2024-01-23T23:16:11Z","snapshot_observed_at":"2026-08-13T10:14:47.864569Z","submitted_at":"2023-09-13T01:07:25Z","title":"Statistical Rejection Sampling Improves Preference Optimization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.06657","snapshot_observed_at":"2026-08-11T23:15:28.677019Z","title":"Statistical rejection sampling improves preference optimization.arXiv preprint arXiv:2309.06657,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.677019Z"},"links":{"cited_paper":"/paper/2309.06657","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:f77b903e29237244920c5a0d827e06a2945aa2b792fe91550855dae2a801d048","observation_id":"8e461d49-9e9e-40c5-a340-1ab0792071c5","resolution":{"observed_at":"2026-08-11T23:15:28.677019Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06592","last_updated":"2024-12-11T22:59:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-05T19:25:40Z","title":"Improve Mathematical Reasoning in Language Models by Automated Process Supervision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06592","snapshot_observed_at":"2026-08-11T23:15:28.682560Z","title":"Improve mathematical reasoning in language models by automated process supervision","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.682560Z"},"links":{"cited_paper":"/paper/2406.06592","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:173341b4ee678facec1749254d16882717f2fd08b14ca8bcb729723dada7bbf4","observation_id":"11592053-4d9d-43a4-880a-94a3c6b9652f","resolution":{"observed_at":"2026-08-11T23:15:28.682560Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.01255","last_updated":"2023-02-17T17:39:41Z","snapshot_observed_at":"2026-08-13T12:42:36.729837Z","submitted_at":"2023-02-17T17:39:41Z","title":"Combining Generative Artificial Intelligence (AI) and the Internet: Heading towards Evolution or Degradation?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.01255","snapshot_observed_at":"2026-08-11T23:15:28.686900Z","title":"Combining generative artificial intelligence (ai) and the internet: Heading towards evolution or degradation? arXiv preprint arXiv:2303.01255,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.686900Z"},"links":{"cited_paper":"/paper/2303.01255","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:9359b4ecaf13614c914b5490fd8a281e8141e0b76319078afe71f546977ce714","observation_id":"3f23aae0-c249-476c-bafe-64b6086acdcb","resolution":{"observed_at":"2026-08-11T23:15:28.686900Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.19733","last_updated":"2024-06-26T01:28:35Z","snapshot_observed_at":"2026-08-13T03:45:12.203086Z","submitted_at":"2024-04-30T17:28:05Z","title":"Iterative Reasoning Preference Optimization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.19733","snapshot_observed_at":"2026-08-11T23:15:28.691595Z","title":"Iterative reasoning preference optimization.arXiv preprint arXiv:2404.19733,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.691595Z"},"links":{"cited_paper":"/paper/2404.19733","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:43c534a07167985c095e7680e41962b95b678b548dffac96a7d06d12505a9731","observation_id":"4ff5c2a0-a54e-467e-a1bf-9bb59e954622","resolution":{"observed_at":"2026-08-11T23:15:28.691595Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.10938","last_updated":"2024-10-01T23:38:10Z","snapshot_observed_at":"2026-08-13T00:04:55.312158Z","submitted_at":"2024-05-17T17:49:44Z","title":"Observational Scaling Laws and the Predictability of Language Model Performance","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.10938","snapshot_observed_at":"2026-08-11T23:15:28.695929Z","title":"Observational scaling laws and the predictability of language model performance.arXiv preprint arXiv:2405.10938,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.695929Z"},"links":{"cited_paper":"/paper/2405.10938","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:0c6b57675e989b62134b7862d1f841dcc0bd267f094d75c91943c1d5a9320ee6","observation_id":"2969d4bf-fb2b-480c-aab9-9f3cccf6ecd0","resolution":{"observed_at":"2026-08-11T23:15:28.695929Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.06585","last_updated":"2024-04-18T03:12:09Z","snapshot_observed_at":"2026-08-13T05:04:53.813940Z","submitted_at":"2023-12-11T18:17:43Z","title":"Beyond Human Data: Scaling Self-Training for Problem-Solving with Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.06585","snapshot_observed_at":"2026-08-11T23:15:28.701530Z","title":"Beyond human data: Scaling self-training for problem-solving with language models.arXiv preprint arXiv:2312.06585,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.701530Z"},"links":{"cited_paper":"/paper/2312.06585","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:9a38f4dbdc62bc2f69b1e51ba8a983ea0427c14504eec92bf09d054dd1fdda3c","observation_id":"bd27e811-fe89-4086-94dc-46179857da5f","resolution":{"observed_at":"2026-08-11T23:15:28.701530Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03314","last_updated":"2024-08-06T17:35:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:35:05Z","title":"Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03314","snapshot_observed_at":"2026-08-11T23:15:28.707967Z","title":"Scaling llm test-time compute optimally can be more effective than scaling model parameters.arXiv preprint arXiv:2408.03314,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.707967Z"},"links":{"cited_paper":"/paper/2408.03314","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:883bac97eef04a7e41c953bcd02efddeeb0c3222070fe34d27e3f3e08a2b332a","observation_id":"f2b82ff6-6639-41aa-bc6c-a542e42df54e","resolution":{"observed_at":"2026-08-11T23:15:28.707967Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15020","last_updated":"2023-07-27T17:24:09Z","snapshot_observed_at":"2026-08-13T15:58:02.303851Z","submitted_at":"2023-07-27T17:24:09Z","title":"SuperCLUE: A Comprehensive Chinese Large Language Model Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15020","snapshot_observed_at":"2026-08-11T23:15:28.712803Z","title":"Moss: Training conversational language models from synthetic data","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.712803Z"},"links":{"cited_paper":"/paper/2307.15020","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:fc754460b76d02b9875255b0862941f7579b25577c46323af8ad0ff73f19f2cd","observation_id":"0e46a5ea-b307-4b1d-bed4-778114a6a13d","resolution":{"observed_at":"2026-08-11T23:15:28.712803Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-11T23:15:28.722455Z","title":"Gemini: a family of highly capable multimodal models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.722455Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:c1b5c7505ab0ae05f26de6143e4244d159caab80cd8244d1ebaf351326f8653a","observation_id":"31de1885-8f17-4558-ab61-91933eaa087e","resolution":{"observed_at":"2026-08-11T23:15:28.722455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-11T23:15:28.726482Z","title":"Llama 2: Open foundation and fine-tuned chat models.arXiv preprint arXiv:2307.09288,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.726482Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:71e2dc4d9f31ef488fa22a5bc64d11fc4e01a8fe922105056fcb85d67a7ab5f3","observation_id":"2354056d-55f4-4c98-a313-2a42c94dcf98","resolution":{"observed_at":"2026-08-11T23:15:28.726482Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.02666","last_updated":"2024-08-08T17:09:58Z","snapshot_observed_at":"2026-08-12T23:08:15.634578Z","submitted_at":"2024-08-05T17:57:02Z","title":"Self-Taught Evaluators","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.02666","snapshot_observed_at":"2026-08-11T23:15:28.731425Z","title":"Math-shepherd: Verify and reinforce llms step-by-step without human annotations","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.731425Z"},"links":{"cited_paper":"/paper/2408.02666","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:641fd93d0da44c2956fe850499e52fa2bb0edcd786613687568cb3e4301279f9","observation_id":"b0d9e045-3538-42cf-9d5c-60c33672daa0","resolution":{"observed_at":"2026-08-11T23:15:28.731425Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.16838","last_updated":"2024-11-20T17:57:26Z","snapshot_observed_at":"2026-08-12T23:35:52.953948Z","submitted_at":"2024-06-24T17:45:59Z","title":"From Decoding to Meta-Generation: Inference-time Algorithms for Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.16838","snapshot_observed_at":"2026-08-11T23:15:28.742927Z","title":"From decoding to meta-generation: Inference-time algorithms for large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.742927Z"},"links":{"cited_paper":"/paper/2406.16838","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:5044e2347c8b4e6ffe3c59323bcaa0b725abf58179e1e7b389e26cc9db2fc815","observation_id":"1fb67c28-3388-46d9-b612-cde357028e3a","resolution":{"observed_at":"2026-08-11T23:15:28.742927Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00724","last_updated":"2025-03-03T07:53:32Z","snapshot_observed_at":"2026-08-13T03:06:10.986532Z","submitted_at":"2024-08-01T17:16:04Z","title":"Inference Scaling Laws: An Empirical Analysis of Compute-Optimal Inference for Problem-Solving with Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00724","snapshot_observed_at":"2026-08-11T23:15:28.747381Z","title":"An empirical analysis of compute- optimal inference for problem-solving with language models.arXiv preprint arXiv:2408.00724,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.747381Z"},"links":{"cited_paper":"/paper/2408.00724","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:a832f9e1f7bf1a130ff7a6e6393ed77ae8bd8f4a424146616feb45806330df54","observation_id":"98b2924e-d1d3-456d-bba8-9439cf7f9261","resolution":{"observed_at":"2026-08-11T23:15:28.747381Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10671","last_updated":"2024-09-10T13:25:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-15T12:35:42Z","title":"Qwen2 Technical Report","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10671","snapshot_observed_at":"2026-08-11T23:15:28.751851Z","title":"Qwen2 technical report.arXiv preprint arXiv:2407.10671,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.751851Z"},"links":{"cited_paper":"/paper/2407.10671","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:cfad9ea98a95aef79f95683d1dd0c7d1d9a00d963e782a647b6f54a278b7c32c","observation_id":"9b3c2369-6c8b-499e-866a-88410e6e5615","resolution":{"observed_at":"2026-08-11T23:15:28.751851Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.14367","last_updated":"2024-01-25T18:14:57Z","snapshot_observed_at":"2026-08-13T04:34:35.372204Z","submitted_at":"2024-01-25T18:14:57Z","title":"Genie: Achieving Human Parity in Content-Grounded Datasets Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.14367","snapshot_observed_at":"2026-08-11T23:15:28.755772Z","title":"Genie: Achieving human parity in content-grounded datasets generation.arXiv preprint arXiv:2401.14367,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.755772Z"},"links":{"cited_paper":"/paper/2401.14367","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:54dcf269af42ce239eccd9a389e70cb7e4a19d0de0b49527ee8a34b334d9ed6f","observation_id":"4dc7458b-b4c8-4da7-b51e-d960c619ba71","resolution":{"observed_at":"2026-08-11T23:15:28.755772Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04652","last_updated":"2025-01-21T10:12:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-07T16:52:49Z","title":"Yi: Open Foundation Models by 01.AI","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04652","snapshot_observed_at":"2026-08-11T23:15:28.760232Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.760232Z"},"links":{"cited_paper":"/paper/2403.04652","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:4bd151bb0f201f5fa56927b746636d77fddcb987fa1b0ea6cf2dc433e8eb72d6","observation_id":"629bf445-2d55-4cc5-83a5-7cabcce69417","resolution":{"observed_at":"2026-08-11T23:15:28.760232Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.10020","last_updated":"2025-03-28T00:06:51Z","snapshot_observed_at":"2026-08-07T08:02:34.857823Z","submitted_at":"2024-01-18T14:43:47Z","title":"Self-Rewarding Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.10020","snapshot_observed_at":"2026-08-11T23:15:28.764847Z","title":"Self-rewarding language models.arXiv preprint arXiv:2401.10020,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.764847Z"},"links":{"cited_paper":"/paper/2401.10020","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:4625eb946624e882322dc3e705a293bee45ac566d546291a6dcb0bdc6c666493","observation_id":"0489c865-ea9d-4798-8599-1d8488769214","resolution":{"observed_at":"2026-08-11T23:15:28.764847Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.09629","last_updated":"2024-03-18T07:56:48Z","snapshot_observed_at":"2026-07-31T09:25:34.205362Z","submitted_at":"2024-03-14T17:58:16Z","title":"Quiet-STaR: Language Models Can Teach Themselves to Think Before Speaking","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.09629","snapshot_observed_at":"2026-08-11T23:15:28.769802Z","title":"Quiet-star: Language models can teach themselves to think before speaking.arXiv preprint arXiv:2403.09629,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.769802Z"},"links":{"cited_paper":"/paper/2403.09629","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:8c572ddb4a03391fd75aad9cb6e3029500bb4e3cf67e46d9a6ad8d9879e8a45e","observation_id":"d00dfe03-37cc-417d-b210-f49f2e73307d","resolution":{"observed_at":"2026-08-11T23:15:28.769802Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07394","last_updated":"2024-06-13T07:19:06Z","snapshot_observed_at":"2026-08-12T23:45:25.243322Z","submitted_at":"2024-06-11T16:01:07Z","title":"Accessing GPT-4 level Mathematical Olympiad Solutions via Monte Carlo Tree Self-refine with LLaMa-3 8B","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.07394","snapshot_observed_at":"2026-08-11T23:15:28.775413Z","title":"Accessing gpt-4 level mathematical olympiad solutions via monte carlo tree self-refine with llama-3 8b.arXiv preprint arXiv:2406.07394, 2024a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.775413Z"},"links":{"cited_paper":"/paper/2406.07394","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:77931d17e65c773b41b9e254e18492e78ab7128d97bb629e3e3cb88b217fee60","observation_id":"143828bb-371f-43d0-8ba7-0e6518bb23a7","resolution":{"observed_at":"2026-08-11T23:15:28.775413Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:15:29.842370Z","title":null,"venue":null,"work_id":"689ae910-d103-41b2-ac88-71f8a4843bf8","year":2023},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.783877Z"},"links":{"citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:59577d084188c806df03ef67bcab818cd256012b58be23ee04e57ff0ce0f7915","observation_id":"e47622b5-ed4e-465e-a80b-cfeee8c31b99","resolution":{"observed_at":"2026-08-11T23:15:29.846814Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:15:29.830402Z","title":null,"venue":null,"work_id":"367ed3e3-62ac-4dbe-b0a9-68c591e8bfa5","year":2000},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.788743Z"},"links":{"citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:35ce364eb8724399c7e83e801185890f77c06eb5963ebe8942e1fb393fc1909f","observation_id":"3c1fe777-9ef9-46c5-9fea-6da85f5d4fbb","resolution":{"observed_at":"2026-08-11T23:15:29.834256Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:15:29.815259Z","title":"The answer is ANSWER","venue":null,"work_id":"3cd81baf-3625-4f8b-a8ff-ad73d03495e3","year":2000},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.793471Z"},"links":{"citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:5df497d7e6fe546b5da27c8c71cf14d0d110d074ea1df366a19a0503275110d5","observation_id":"9b431a19-324a-48eb-af09-6baf75bbfade","resolution":{"observed_at":"2026-08-11T23:15:29.821481Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.04707","last_updated":"2024-10-07T02:52:30Z","snapshot_observed_at":"2026-08-13T19:57:51.671184Z","submitted_at":"2024-10-07T02:52:30Z","title":"Learning How Hard to Think: Input-Adaptive Allocation of LM Computation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.04707","snapshot_observed_at":"2026-08-11T23:15:28.575938Z","title":"Learning how hard to think: Input-adaptive allocation of lm computation.arXiv preprint arXiv:2410.04707,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":2005,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.575938Z"},"links":{"cited_paper":"/paper/2410.04707","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:3ad65e2eb8f185baf7e7e2ab3ab78c08b7198a715677d4f5e2ce31c5299c6491","observation_id":"4bb79ffe-ae45-4439-87d8-7c3a80a411ff","resolution":{"observed_at":"2026-08-11T23:15:28.575938Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.11704","last_updated":"2025-05-23T16:32:54Z","snapshot_observed_at":"2026-08-13T20:13:18.325386Z","submitted_at":"2024-09-18T05:13:18Z","title":"From Lists to Emojis: How Format Bias Affects Model Alignment","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.11704","snapshot_observed_at":"2026-08-11T23:15:28.779769Z","title":"From lists to emojis: How format bias affects model alignment.arXiv preprint arXiv:2409.11704, 2024c","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":2006,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.779769Z"},"links":{"cited_paper":"/paper/2409.11704","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:94da9f9bfd414b60d9e5e37c6820fd2efb3ccc0db9cf038eddbb226c6adb9afc","observation_id":"1faa3f38-df3a-46fd-b810-1c80e0fae732","resolution":{"observed_at":"2026-08-11T23:15:28.779769Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12186","last_updated":"2024-11-12T13:24:25Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-18T17:57:57Z","title":"Qwen2.5-Coder Technical Report","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12186","snapshot_observed_at":"2026-08-11T23:15:28.633561Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":2007,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.633561Z"},"links":{"cited_paper":"/paper/2409.12186","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:a688cb69ad3d5564a753e9969a1b94621391440e1918a999f890a18418fa8ded","observation_id":"63d336fa-0c4c-4805-8f21-c9204165c948","resolution":{"observed_at":"2026-08-11T23:15:28.633561Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03874","last_updated":"2021-11-08T21:30:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-03-05T18:59:39Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.03874","snapshot_observed_at":"2026-08-11T23:15:28.619469Z","title":"Measuring mathematical problem solving with the math dataset","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":2016,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.619469Z"},"links":{"cited_paper":"/paper/2103.03874","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:cbe6a6a992d873d3e0ea36dc09bebff3b464fa11fb8b4745ad6f9bb76d5ecbf8","observation_id":"1b028eb2-10ef-4b77-ba70-6096fc73c12b","resolution":{"observed_at":"2026-08-11T23:15:28.619469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-11T23:15:28.571867Z","title":"Training verifiers to solve math word problems","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.571867Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:055c83e59e419cb2426ac5bd14760ca2b7fa26b1897b4c53ed08ea609bb5fa98","observation_id":"792c3e5f-31ff-46e0-944c-8e18e8bec09d","resolution":{"observed_at":"2026-08-11T23:15:28.571867Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.09726","last_updated":"2022-11-14T19:22:02Z","snapshot_observed_at":"2026-08-13T15:40:33.656736Z","submitted_at":"2022-05-19T17:36:46Z","title":"RankGen: Improving Text Generation with Large Ranking Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.09726","snapshot_observed_at":"2026-08-11T23:15:28.654641Z","title":"Rankgen: Improving text generation with large ranking models.arXiv preprint arXiv:2205.09726,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.654641Z"},"links":{"cited_paper":"/paper/2205.09726","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:ee51191a262e22887738d425dc012a3997c1e139465cbc251c4b819a996b56c5","observation_id":"1064790f-6c2e-435d-a6bf-1a55ab4286f9","resolution":{"observed_at":"2026-08-11T23:15:28.654641Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2102.01293","last_updated":"2021-02-02T04:07:38Z","snapshot_observed_at":"2026-08-01T22:46:19.170916Z","submitted_at":"2021-02-02T04:07:38Z","title":"Scaling Laws for Transfer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2102.01293","snapshot_observed_at":"2026-08-11T23:15:28.624117Z","title":"Scaling laws for transfer.arXiv preprint arXiv:2102.01293,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.624117Z"},"links":{"cited_paper":"/paper/2102.01293","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:af708807c59219d4bfbb00d62fa335ddcb0e1c81e70c2dc8420171c074aeac43","observation_id":"243fad25-7fbf-4e43-ad28-32c556fbd12c","resolution":{"observed_at":"2026-08-11T23:15:28.624117Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.16737","last_updated":"2024-10-07T19:37:10Z","snapshot_observed_at":"2026-08-12T22:54:59.515243Z","submitted_at":"2024-08-29T17:32:35Z","title":"Smaller, Weaker, Yet Better: Training LLM Reasoners via Compute-Optimal Sampling","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.16737","snapshot_observed_at":"2026-08-11T23:15:28.520053Z","title":"Smaller, weaker, yet better: Training llm reasoners via compute-optimal sampling.arXiv preprint arXiv:2408.16737,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.520053Z"},"links":{"cited_paper":"/paper/2408.16737","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:273542ff15cf909712cc3073a2928dd006ff7d486c824110bb418f7f987534d0","observation_id":"96760804-8764-4cbe-b70a-66b31af334d1","resolution":{"observed_at":"2026-08-11T23:15:28.520053Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11704","last_updated":"2024-08-06T22:37:06Z","snapshot_observed_at":"2026-08-12T23:40:54.718397Z","submitted_at":"2024-06-17T16:25:04Z","title":"Nemotron-4 340B Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11704","snapshot_observed_at":"2026-08-11T23:15:28.488650Z","title":"Nemotron-4 340b technical report.arXiv preprint arXiv:2406.11704,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.488650Z"},"links":{"cited_paper":"/paper/2406.11704","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:b1f5b898b18b9fcf53d2400f4306a77b5688f3bae4a50eea78662db7e2f496b5","observation_id":"a423a56c-b440-4db2-9426-3ce9500910d7","resolution":{"observed_at":"2026-08-11T23:15:28.488650Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.01850","last_updated":"2023-07-04T17:59:31Z","snapshot_observed_at":"2026-08-13T11:02:52.030577Z","submitted_at":"2023-07-04T17:59:31Z","title":"Self-Consuming Generative Models Go MAD","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.01850","snapshot_observed_at":"2026-08-11T23:15:28.493309Z","title":"Self-consuming generative models go mad.arXiv preprint arXiv:2307.01850,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.493309Z"},"links":{"cited_paper":"/paper/2307.01850","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:e5a86c24dab21ed4aad0e72de8e3cf28531588e515971732860c2cbfaeac12cd","observation_id":"c3c60f7e-5b8b-448f-81b5-2d195bebe09b","resolution":{"observed_at":"2026-08-11T23:15:28.493309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.11610","last_updated":"2022-10-25T17:45:17Z","snapshot_observed_at":"2026-08-09T02:27:34.152063Z","submitted_at":"2022-10-20T21:53:54Z","title":"Large Language Models Can Self-Improve","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.11610","snapshot_observed_at":"2026-08-11T23:15:28.629098Z","title":"Jiaxin Huang, Shixiang Shane Gu, Le Hou, Yuexin Wu, Xuezhi Wang, Hongkun Yu, and Jiawei Han","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-11T23:15:28.629098Z"},"links":{"cited_paper":"/paper/2210.11610","citing_paper":"/paper/2412.02674"},"observation_digest":"sha256:7cdde1b8bdb4a413dbaca3284ef188a7d15ab7c3602f20dff094b2f13fb8c64c","observation_id":"1480e59d-4c59-483d-8432-dd0882773667","resolution":{"observed_at":"2026-08-11T23:15:28.629098Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-11T23:09:04.859107Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models"},"reference_resolution":{"displayed":62,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":61,"verified_exact":0,"verified_fuzzy":1},"total_outbound_references":62},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 62 of 62 outbound references and 18 inbound Pith citation observations for arXiv:2412.02674."}