{"as_of":"2026-08-10T21:58:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:beabf5340d0966a995f97f5c6c66e8e8d13b720682e71bfc592af9685cdcb84b","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":29,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":29,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":29,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":29,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T20:42:44.023438Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T06:59:38.178795Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":"2405.03553","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-07-04T06:59:38.178795Z","title":"Alphamath almost zero: process supervision without process","venue":null,"work_id":"142e2ffe-057a-4f74-9691-31fc3b21fb03","year":2024},"citing_paper":{"arxiv_id":"2406.18629","last_updated":"2024-06-26T17:43:06Z","snapshot_observed_at":"2026-08-06T00:24:52.274888Z","submitted_at":"2024-06-26T17:43:06Z","title":"Step-DPO: Step-wise Preference Optimization for Long-chain Reasoning of LLMs","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-18T23:58:29.040819Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2406.18629"},"observation_digest":"sha256:2bb84a08680b997e163e27e1644552f5561271babc3371e60741e0a049222751","observation_id":"95df9dbc-b51e-4716-84d4-a1ff874671d7","resolution":{"observed_at":"2026-05-18T23:58:29.092722Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":"2405.03553","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-07-04T06:59:38.178795Z","title":"Alphamath almost zero: process supervision without process","venue":null,"work_id":"142e2ffe-057a-4f74-9691-31fc3b21fb03","year":2024},"citing_paper":{"arxiv_id":"2501.07301","last_updated":"2025-06-05T16:34:24Z","snapshot_observed_at":"2026-08-03T11:11:25.359494Z","submitted_at":"2025-01-13T13:10:16Z","title":"The Lessons of Developing Process Reward Models in Mathematical Reasoning","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-16T13:43:43.223872Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2501.07301"},"observation_digest":"sha256:77468068cf9c7f216bd811e4930c0d0a5f7ddea64c1a041e7d82c9b6ced08517","observation_id":"4e7de639-b7b9-4499-b210-0a96af1fd149","resolution":{"observed_at":"2026-05-16T13:43:43.262923Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":"2405.03553","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-07-04T06:59:38.178795Z","title":"Alphamath almost zero: process supervision without process","venue":null,"work_id":"142e2ffe-057a-4f74-9691-31fc3b21fb03","year":2024},"citing_paper":{"arxiv_id":"2501.09686","last_updated":"2025-01-23T08:44:44Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-16T17:37:58Z","title":"Towards Large Reasoning Models: A Survey of Reinforced Reasoning with Large Language Models","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-15T21:20:59.128986Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2501.09686"},"observation_digest":"sha256:a5c14ea9b853af20e7681262db995d79a47002a4b4bd509425c21001149d5ed2","observation_id":"aba8b193-025b-4266-b179-838f641dfc04","resolution":{"observed_at":"2026-05-15T21:20:59.482839Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":"2405.03553","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-07-04T06:59:38.178795Z","title":"Alphamath almost zero: process supervision without process","venue":null,"work_id":"142e2ffe-057a-4f74-9691-31fc3b21fb03","year":2024},"citing_paper":{"arxiv_id":"2501.17161","last_updated":"2025-05-26T17:16:45Z","snapshot_observed_at":"2026-08-09T18:26:12.869738Z","submitted_at":"2025-01-28T18:59:44Z","title":"SFT Memorizes, RL Generalizes: A Comparative Study of Foundation Model Post-training","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-12T21:31:30.202477Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2501.17161"},"observation_digest":"sha256:b1a97bd9a01b6365fc730072794dba2cd12415aee7b9b164c28fcb89c53aa15f","observation_id":"d942bd42-d465-408f-8559-ec408d011617","resolution":{"observed_at":"2026-05-12T21:31:30.251450Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-08-09T20:42:44.023438Z","title":"Alphamath almost zero: process supervision without process","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.19324","last_updated":"2025-06-26T03:14:46Z","snapshot_observed_at":"2026-08-10T13:46:16.822794Z","submitted_at":"2025-01-31T17:19:57Z","title":"Reward-Guided Speculative Decoding for Efficient LLM Reasoning","version":3},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-09T20:42:44.023438Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2501.19324"},"observation_digest":"sha256:301cd3bec415a3b856d93e6e4a042766422280d5125112ea54ec10b5190dd47f","observation_id":"af617fc0-5954-4655-8222-a5b29688ce9e","resolution":{"observed_at":"2026-08-09T20:42:44.023438Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-08-09T19:40:23.078834Z","title":"Alphamath almost zero: process supervision without process","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.00271","last_updated":"2025-02-01T02:08:49Z","snapshot_observed_at":"2026-08-09T19:33:40.106138Z","submitted_at":"2025-02-01T02:08:49Z","title":"Scaling Flaws of Verifier-Guided Search in Mathematical Reasoning","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-09T19:40:23.078834Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2502.00271"},"observation_digest":"sha256:6f9889630a7bc122df225602577882d0bf5eb49a512274b2ab78c4bddb3749bf","observation_id":"85ff0d55-9264-4da6-917e-1d6b3c42b054","resolution":{"observed_at":"2026-08-09T19:40:23.078834Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-08-09T13:25:51.975606Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.02095","last_updated":"2025-05-20T12:35:46Z","snapshot_observed_at":"2026-08-10T02:46:24.418647Z","submitted_at":"2025-02-04T08:25:17Z","title":"LongDPO: Unlock Better Long-form Generation Abilities for LLMs via Critique-augmented Stepwise Information","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-09T13:25:51.975606Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2502.02095"},"observation_digest":"sha256:25fee00adffba964641e7f64ceac390e3732c3a640aa4f66c5b1373233118277","observation_id":"3aaf91ed-6d16-4599-89f9-4da1e791b91a","resolution":{"observed_at":"2026-08-09T13:25:51.975606Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-08-08T21:40:26.778702Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.04751","last_updated":"2025-02-07T08:36:39Z","snapshot_observed_at":"2026-08-09T03:57:00.818497Z","submitted_at":"2025-02-07T08:36:39Z","title":"Holistically Guided Monte Carlo Tree Search for Intricate Information Seeking","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-08T21:40:26.778702Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2502.04751"},"observation_digest":"sha256:5859ca0b629177f10ad25998d04e2ec20b14d5ffd03803bbe2099ad6803e907b","observation_id":"7f524a81-af53-4db3-a350-7734de9c90ef","resolution":{"observed_at":"2026-08-08T21:40:26.778702Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-08-08T19:23:26.279454Z","title":"findings-emnlp.463","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.05449","last_updated":"2025-06-01T20:14:16Z","snapshot_observed_at":"2026-08-10T20:57:26.493297Z","submitted_at":"2025-02-08T04:39:51Z","title":"Iterative Deepening Sampling as Efficient Test-Time Scaling","version":2},"reference_index":463,"source":"pdf_text","source_observed_at":"2026-08-08T19:23:26.279454Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2502.05449"},"observation_digest":"sha256:821b2af09f74d3a48c50d09acf0af4c7c71d9d89f41608fbf02d5853c6c78881","observation_id":"cea037dc-7b08-46f4-8717-cedc7b7d32eb","resolution":{"observed_at":"2026-08-08T19:23:26.279454Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-08-08T18:10:53.351858Z","title":"Alphamath almost zero: process supervision without process","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.05773","last_updated":"2025-07-24T22:59:42Z","snapshot_observed_at":"2026-08-10T01:49:25.494930Z","submitted_at":"2025-02-09T04:31:30Z","title":"PIPA: Preference Alignment as Prior-Informed Statistical Estimation","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-08T18:10:53.351858Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2502.05773"},"observation_digest":"sha256:62994adb8b8d7ae235b73b5da369662a8132f8ce994ee8c3d6098f1505fbc991","observation_id":"db981569-bbae-4bbb-9a6c-4b523ed8decb","resolution":{"observed_at":"2026-08-08T18:10:53.351858Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-08-08T14:25:53.308231Z","title":"Alphamath almost zero: process supervision without process","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06773","last_updated":"2025-02-10T18:52:04Z","snapshot_observed_at":"2026-08-10T11:31:53.867683Z","submitted_at":"2025-02-10T18:52:04Z","title":"On the Emergence of Thinking in LLMs I: Searching for the Right Intuition","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-08T14:25:53.308231Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2502.06773"},"observation_digest":"sha256:a1a091599a279ec2debd9b0e9132340720a23bf9b39776ec125ff0cea4a1512d","observation_id":"2402dfe1-17ca-45d9-8aa6-e55610864943","resolution":{"observed_at":"2026-08-08T14:25:53.308231Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":"2405.03553","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-07-04T06:59:38.178795Z","title":"Alphamath almost zero: process supervision without process","venue":null,"work_id":"142e2ffe-057a-4f74-9691-31fc3b21fb03","year":2024},"citing_paper":{"arxiv_id":"2502.17419","last_updated":"2025-06-25T02:24:46Z","snapshot_observed_at":"2026-08-10T09:42:17.185681Z","submitted_at":"2025-02-24T18:50:52Z","title":"From System 1 to System 2: A Survey of Reasoning Large Language Models","version":6},"reference_index":207,"source":"pdf_text","source_observed_at":"2026-05-13T01:36:23.845366Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2502.17419"},"observation_digest":"sha256:19d23a5ee5da923a9ead6ced42358f5daddfc897f38aa2be67f9aa6e9815782d","observation_id":"72aa303b-4f83-4ebd-941d-978a51fbb63b","resolution":{"observed_at":"2026-05-13T01:36:24.317561Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-08-07T15:38:45.934365Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14391","last_updated":"2025-05-20T14:12:05Z","snapshot_observed_at":"2026-08-10T16:59:40.688883Z","submitted_at":"2025-05-20T14:12:05Z","title":"Beyond the First Error: Process Reward Models for Reflective Mathematical Reasoning","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T15:38:45.934365Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2505.14391"},"observation_digest":"sha256:85f791e206bcbb2550b145256b6360ce9f383126f81fce83c6bf5f05756e4590","observation_id":"e6f6a7e9-f55a-4881-9803-c1929b3d812e","resolution":{"observed_at":"2026-08-07T15:38:45.934365Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-08-07T14:42:27.432302Z","title":"Alphamath almost zero: process supervision without process.arXiv preprint arXiv:2405.03553, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.18065","last_updated":"2025-05-23T16:12:12Z","snapshot_observed_at":"2026-08-08T14:50:33.225594Z","submitted_at":"2025-05-23T16:12:12Z","title":"Reward Model Generalization for Compute-Aware Test-Time Reasoning","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T14:42:27.432302Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2505.18065"},"observation_digest":"sha256:40d6efbf8f638f44e7fd0cb72a60f743444241dd5b6653522eeb79ca75eb7049","observation_id":"5deda4ba-41c6-44cd-8c92-812854e08f0d","resolution":{"observed_at":"2026-08-07T14:42:27.432302Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-08-07T14:11:52.046023Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19683","last_updated":"2025-05-26T08:44:53Z","snapshot_observed_at":"2026-08-08T09:25:43.689145Z","submitted_at":"2025-05-26T08:44:53Z","title":"Large Language Models for Planning: A Comprehensive and Systematic Survey","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T14:11:52.046023Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2505.19683"},"observation_digest":"sha256:304b25988b16f3bde4fca14c915aed28753e0da03548b817bcd5e3253e383559","observation_id":"9b4114fc-0234-43e5-b9a5-0942a0a263ee","resolution":{"observed_at":"2026-08-07T14:11:52.046023Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-08-07T14:08:09.356293Z","title":"CoRR, abs/2405.03553","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19997","last_updated":"2025-08-09T05:41:36Z","snapshot_observed_at":"2026-08-07T13:59:31.661131Z","submitted_at":"2025-05-26T13:48:49Z","title":"Embracing Imperfection: Simulating Students with Diverse Cognitive Levels Using LLM-based Agents","version":2},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-07T14:08:09.356293Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2505.19997"},"observation_digest":"sha256:2f52ba4e7bc2a30c5b7de4bc71604b6f9a684d07083444ea9e943ffb1074ca1b","observation_id":"562b797c-8f13-438d-bcde-0e2872195c49","resolution":{"observed_at":"2026-08-07T14:08:09.356293Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-08-07T13:34:12.576689Z","title":"Alphamath almost zero: process supervision without process.arXiv preprint arXiv:2405.03553, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21496","last_updated":"2025-05-27T17:58:06Z","snapshot_observed_at":"2026-08-08T15:12:37.593143Z","submitted_at":"2025-05-27T17:58:06Z","title":"UI-Genie: A Self-Improving Approach for Iteratively Boosting MLLM-based Mobile GUI Agents","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T13:34:12.576689Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2505.21496"},"observation_digest":"sha256:342b5ac18c5b98c08b19d5c02530f7439b8c16060ac6a96189007476801ed41c","observation_id":"0d1cd788-394a-43d5-bbad-e5da891f9fc6","resolution":{"observed_at":"2026-08-07T13:34:12.576689Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-08-07T05:36:38.439633Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07551","last_updated":"2025-06-12T07:30:27Z","snapshot_observed_at":"2026-08-07T22:38:32.445079Z","submitted_at":"2025-06-09T08:41:39Z","title":"CheMatAgent: Enhancing LLMs for Chemistry and Materials Science through Tree-Search Based Tool Learning","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T05:36:38.439633Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2506.07551"},"observation_digest":"sha256:05992838bc8a4945835abd2bbaba1e8adee12df34e4e8bab7eae01701a9e005e","observation_id":"7fb1b3b2-3966-48c9-8a02-100d9ca0a134","resolution":{"observed_at":"2026-08-07T05:36:38.439633Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-08-07T01:08:20.439050Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.11902","last_updated":"2025-06-13T15:52:37Z","snapshot_observed_at":"2026-08-07T00:58:51.821987Z","submitted_at":"2025-06-13T15:52:37Z","title":"TreeRL: LLM Reinforcement Learning with On-Policy Tree Search","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-07T01:08:20.439050Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2506.11902"},"observation_digest":"sha256:e8620a13c163979545e398078daf7b78e1405800d6e8334f1b8df622a708908f","observation_id":"37859dfb-bf4d-40f2-b6a5-eede338bbcf6","resolution":{"observed_at":"2026-08-07T01:08:20.439050Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":"2405.03553","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-07-04T06:59:38.178795Z","title":"Alphamath almost zero: process supervision without process","venue":null,"work_id":"142e2ffe-057a-4f74-9691-31fc3b21fb03","year":2024},"citing_paper":{"arxiv_id":"2507.21046","last_updated":"2026-01-16T20:59:08Z","snapshot_observed_at":"2026-08-01T06:32:44.461162Z","submitted_at":"2025-07-28T17:59:05Z","title":"A Survey of Self-Evolving Agents: What, When, How, and Where to Evolve on the Path to Artificial Super Intelligence","version":4},"reference_index":233,"source":"arxiv_source","source_observed_at":"2026-05-14T22:23:14.621091Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2507.21046"},"observation_digest":"sha256:01358b8c408a6690dcc5156624a2ff9dabc16305b5a978b5baedfe016dc579a9","observation_id":"263e6914-3671-4b38-9243-004647ca830f","resolution":{"observed_at":"2026-05-14T22:23:15.196329Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-08-06T12:53:23.931239Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.21389","last_updated":"2025-07-28T23:50:09Z","snapshot_observed_at":"2026-08-09T15:35:01.078131Z","submitted_at":"2025-07-28T23:50:09Z","title":"Teaching Language Models To Gather Information Proactively","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T12:53:23.931239Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2507.21389"},"observation_digest":"sha256:7f6097b8c9bf06c68b39751aa475d51e0c2e40b2290c13f0f716d8e2c912c0d9","observation_id":"79e7014c-c445-4614-b3f4-abc9e3f31819","resolution":{"observed_at":"2026-08-06T12:53:23.931239Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-08-05T12:25:24.465741Z","title":"Alphamath almost zero: Process supervision without process, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.01649","last_updated":"2025-09-01T17:49:14Z","snapshot_observed_at":"2026-08-09T22:40:15.578714Z","submitted_at":"2025-09-01T17:49:14Z","title":"Distilled Pretraining: A modern lens of Data, In-Context Learning and Test-Time Scaling","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-05T12:25:24.465741Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2509.01649"},"observation_digest":"sha256:8d029cbb5bc17134cc6dc268c0be0b4f0011d34f5f25dbce21d3f198a46f99cd","observation_id":"3751ea62-5e02-48cc-bd39-ce387294c4a2","resolution":{"observed_at":"2026-08-05T12:25:24.465741Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":"2405.03553","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-07-04T06:59:38.178795Z","title":"Alphamath almost zero: process supervision without process","venue":null,"work_id":"142e2ffe-057a-4f74-9691-31fc3b21fb03","year":2024},"citing_paper":{"arxiv_id":"2509.02547","last_updated":"2026-04-17T18:09:08Z","snapshot_observed_at":"2026-08-03T09:07:42.489237Z","submitted_at":"2025-09-02T17:46:26Z","title":"The Landscape of Agentic Reinforcement Learning for LLMs: A Survey","version":5},"reference_index":199,"source":"pdf_text","source_observed_at":"2026-05-18T19:19:36.427337Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2509.02547"},"observation_digest":"sha256:ddc1d0df7b8686f59640f422b2e7f4ba07e07eeae34bbd3d2e88af1e3005036b","observation_id":"65e4ebbc-7384-42e9-ad58-1a12d88c7053","resolution":{"observed_at":"2026-05-18T19:21:48.105505Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-08-05T05:45:02.225374Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.05007","last_updated":"2025-09-08T03:26:03Z","snapshot_observed_at":"2026-08-05T05:45:00.604843Z","submitted_at":"2025-09-05T11:14:11Z","title":"Sticker-TTS: Learn to Utilize Historical Experience with a Sticker-driven Test-Time Scaling Framework","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-05T05:45:02.225374Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2509.05007"},"observation_digest":"sha256:e4ea02049648b57fd86a1d391a3553b837c4c31396114c3fd25c6a721444350a","observation_id":"da57585d-8e1f-4826-b173-bb9b6272ace1","resolution":{"observed_at":"2026-08-05T05:45:02.225374Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-08-03T07:08:32.399343Z","title":"ArXiv:2405.03553","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2601.21169","last_updated":"2026-01-29T02:11:43Z","snapshot_observed_at":"2026-08-08T13:22:13.997719Z","submitted_at":"2026-01-29T02:11:43Z","title":"Output-Space Search: Targeting LLM Generations in a Frozen Encoder-Defined Output Space","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-03T07:08:32.399343Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2601.21169"},"observation_digest":"sha256:6b39fac84a792df8490da990ec7f2a03b6296113f6586838d4bd1c6f8a0af27d","observation_id":"70ecd80a-1fbe-4573-bfbf-687568a7f43c","resolution":{"observed_at":"2026-08-03T07:08:32.399343Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":"2405.03553","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-07-04T06:59:38.178795Z","title":"Alphamath almost zero: process supervision without process","venue":null,"work_id":"142e2ffe-057a-4f74-9691-31fc3b21fb03","year":2024},"citing_paper":{"arxiv_id":"2604.10660","last_updated":"2026-04-12T14:28:35Z","snapshot_observed_at":"2026-07-06T22:59:13.936436Z","submitted_at":"2026-04-12T14:28:35Z","title":"Efficient Process Reward Modeling via Contrastive Mutual Information","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T16:40:04.180754Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2604.10660"},"observation_digest":"sha256:eae4c800e1c1c8484cdd905a7834b777f24e167c07bbea025da5ff46adc2c61f","observation_id":"84b7865a-05ed-4ac7-96ad-f9356ecc90d4","resolution":{"observed_at":"2026-05-11T08:25:59.087302Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":"2405.03553","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-07-04T06:59:38.178795Z","title":"Alphamath almost zero: process supervision without process","venue":null,"work_id":"142e2ffe-057a-4f74-9691-31fc3b21fb03","year":2024},"citing_paper":{"arxiv_id":"2606.05464","last_updated":"2026-06-03T21:43:38Z","snapshot_observed_at":"2026-08-03T10:19:08.677126Z","submitted_at":"2026-06-03T21:43:38Z","title":"Step-by-Step Optimization-like Reasoning in LLMs over Expanding Search Spaces","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-06-28T05:46:26.938277Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2606.05464"},"observation_digest":"sha256:870a6a64e2843f54881337a9436f3b8019642dfa93e284163bec589d6fff0d66","observation_id":"e1732b0c-e87f-47b7-95d4-6a4c3984777f","resolution":{"observed_at":"2026-07-02T08:46:48.970743Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":"2405.03553","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-07-04T06:59:38.178795Z","title":"Alphamath almost zero: process supervision without process","venue":null,"work_id":"142e2ffe-057a-4f74-9691-31fc3b21fb03","year":2024},"citing_paper":{"arxiv_id":"2606.10568","last_updated":"2026-06-09T08:31:59Z","snapshot_observed_at":"2026-07-06T23:49:47.137691Z","submitted_at":"2026-06-09T08:31:59Z","title":"VeriSpace: Spatially Grounded Action Verification for Vision-Language-Action Models","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-27T12:47:35.314486Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2606.10568"},"observation_digest":"sha256:64b34acd79abe479f2dfbe40217bdf242fa444121dc33cd8d16b76953967a545","observation_id":"f0f900f5-924b-43e0-9611-b935c93847ef","resolution":{"observed_at":"2026-07-03T06:17:41.795472Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process","version":3},"cited_work":{"arxiv_id":"2405.03553","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.03553","snapshot_observed_at":"2026-07-04T06:59:38.178795Z","title":"Alphamath almost zero: process supervision without process","venue":null,"work_id":"142e2ffe-057a-4f74-9691-31fc3b21fb03","year":2024},"citing_paper":{"arxiv_id":"2606.21740","last_updated":"2026-06-19T20:53:10Z","snapshot_observed_at":"2026-08-01T23:24:43.869329Z","submitted_at":"2026-06-19T20:53:10Z","title":"Training the Orchestrator: A Supervised Approach to End-to-End PDDL Planning with LLM Agents","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-06-26T13:56:51.914966Z"},"links":{"cited_paper":"/paper/2405.03553","citing_paper":"/paper/2606.21740"},"observation_digest":"sha256:a4e13618a6bf813d8a305ef645396cd1a0400b7448ff0a279696c46c9d12daff","observation_id":"fc201724-9778-4f9a-ba21-49f95a24563e","resolution":{"observed_at":"2026-07-04T06:59:38.180175Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2405.03553/citation-record","integrity":"/paper/2405.03553/integrity","json":"/paper/2405.03553/citation-record.json","paper":"/paper/2405.03553"},"outbound":[],"paper":{"arxiv_id":"2405.03553","last_updated":"2024-09-27T08:16:28Z","latest_version":3,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-08T14:49:36.757543Z","submitted_at":"2024-05-06T15:20:30Z","title":"AlphaMath Almost Zero: Process Supervision without Process"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 29 inbound Pith citation observations for arXiv:2405.03553."}