{"as_of":"2026-08-07T02:29:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d29f7a74f58ab6091266526622369e344de03eddf71c722c13a3f1328e42ba6b","coverage":[{"denominator":82,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":82,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-13T05:02:49.206053Z","state":"measured"},{"denominator":83,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":83,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T07:34:39.055539Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"cited_work":{"arxiv_id":"2605.12004","doi":"10.48550/arxiv.2605.12004","metadata_source":"pith","pith_arxiv_id":"2605.12004","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"Learning Agentic Policy from Action Guidance","venue":"cs.CL","work_id":"cd596862-59c6-4c4d-99ac-c6ed278a2dba","year":2026},"citing_paper":{"arxiv_id":"2607.21419","last_updated":"2026-07-30T03:55:54Z","snapshot_observed_at":"2026-08-07T00:55:35.431397Z","submitted_at":"2026-07-23T15:24:35Z","title":"PATS: Policy-Aware Training Scaffolding for Agentic Reinforcement Learning","version":2},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-01T07:34:39.055539Z"},"links":{"cited_paper":"/paper/2605.12004","citing_paper":"/paper/2607.21419"},"observation_digest":"sha256:6999d40d8cf696987ff43b7f2ab3740a1a43a80514c660e6bd34b1593d702e59","observation_id":"1e1e43b4-438e-4481-bbd8-f32e36544441","resolution":{"observed_at":"2026-08-01T07:39:17.666069Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-01T10:08:10.546089+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T10:08:10.546089+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2605.12004/citation-record","integrity":"/paper/2605.12004/integrity","json":"/paper/2605.12004/citation-record.json","paper":"/paper/2605.12004"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Claude Opus 4.6 model card","venue":null,"work_id":"3bb72bcf-3887-42cb-9f78-ca4e454cad7f","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:0db0bad4e23c8f581b10fa6eca56358812fffab809135f21efa82a38da7ca969","observation_id":"1639d43e-7ea4-4e77-9b48-1f80e58be624","resolution":{"observed_at":"2026-05-13T11:07:39.831018Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.07982","last_updated":"2025-06-09T17:52:18Z","snapshot_observed_at":"2026-07-06T21:39:13.304260Z","submitted_at":"2025-06-09T17:52:18Z","title":"$\\tau^2$-Bench: Evaluating Conversational Agents in a Dual-Control Environment","version":1},"cited_work":{"arxiv_id":"2506.07982","doi":"10.48550/arxiv.2506.07982","metadata_source":"pith","pith_arxiv_id":"2506.07982","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"$\\tau^2$-Bench: Evaluating Conversational Agents in a Dual-Control Environment","venue":"cs.AI","work_id":"3a498b1a-455f-4667-b572-c5216c99a89c","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2506.07982","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:63450e673374cb01ce745ace7dcbad695fc3af48a8f94a64cd5989b72e2449a2","observation_id":"5691e81c-db21-469e-914f-a15cb41526d4","resolution":{"observed_at":"2026-05-13T05:07:17.802510Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:22.129969+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:22.129969+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Fine- tuning web agents: It works, but it’s trickier than you think","venue":null,"work_id":"27825400-9f46-4f37-a856-c78af8d6001d","year":2024},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:947c76dbab0f1dec3f855f7d44be73bdc5089cba7a49e64035ea2f66442389a1","observation_id":"cf10f969-fe6e-4803-be0b-9cd962e0eda4","resolution":{"observed_at":"2026-05-13T11:07:39.816125Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.11468","last_updated":"2025-04-10T16:54:05Z","snapshot_observed_at":"2026-08-05T02:57:13.928237Z","submitted_at":"2025-04-10T16:54:05Z","title":"SFT or RL? An Early Investigation into Training R1-Like Reasoning Large Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2504.11468","doi":"10.48550/arxiv.2504.11468","metadata_source":"pith","pith_arxiv_id":"2504.11468","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SFT or RL? An Early Investigation into Training R1-Like Reasoning Large Vision-Language Models","venue":"cs.CL","work_id":"a521360c-8673-4d0d-a3a3-6eb9f7a71b90","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2504.11468","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:f4d98e490e76e5829561df821bf89f77e883a25c4b0538aa275f5c6a67c8f5d4","observation_id":"f2176436-10eb-4ea0-bbf8-da2b115a9287","resolution":{"observed_at":"2026-05-17T15:43:34.765363Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13651","last_updated":"2025-06-16T16:16:14Z","snapshot_observed_at":"2026-08-07T00:25:54.557015Z","submitted_at":"2025-06-16T16:16:14Z","title":"xbench: Tracking Agents Productivity Scaling with Profession-Aligned Real-World Evaluations","version":1},"cited_work":{"arxiv_id":"2506.13651","doi":"10.48550/arxiv.2506.13651","metadata_source":"pith","pith_arxiv_id":"2506.13651","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"xbench: Tracking agents productivity scaling with profession-aligned real-world evaluations","venue":"cs.LG","work_id":"b7f0bdf6-3821-4735-b55e-be58f6ef326d","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2506.13651","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:e0eb364f76a9fc3121e70156c60e199337238942668f2b96424e96ed7a88b6af","observation_id":"a2d792c1-5c8e-444f-8ae2-27ecb873b73a","resolution":{"observed_at":"2026-05-13T05:07:17.799791Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.06948","last_updated":"2026-05-30T09:01:09Z","snapshot_observed_at":"2026-08-04T22:56:01.963927Z","submitted_at":"2025-09-08T17:58:02Z","title":"Beyond Two-Stage Training: Cooperative SFT and RL for LLM Reasoning","version":3},"cited_work":{"arxiv_id":"2509.06948","doi":"10.48550/arxiv.2509.06948","metadata_source":"pith","pith_arxiv_id":"2509.06948","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Beyond two-stage training: Cooperative sft and rl for llm reasoning","venue":"cs.CL","work_id":"c187c8ff-10ae-4b69-9b4d-8a04c661f929","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2509.06948","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:baba4f90e03bbedc5c8fc6449f0064f3b936075689161678474737df9405638a","observation_id":"357d6141-dc4b-4d87-8a14-a127566deef6","resolution":{"observed_at":"2026-06-02T02:03:33.109892Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T09:34:48.462851Z","title":"GPG: A simple and strong reinforcement learning baseline for model reasoning","venue":null,"work_id":"97c1efeb-7277-42d6-b9c1-53580a47bd15","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:55dba615d5a5eac6b587f5042d7697ce8bda34b710b876791122c1e72f255492","observation_id":"987e40b8-e036-4fc2-85eb-758838d08a28","resolution":{"observed_at":"2026-05-13T11:07:39.825455Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.14234","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T20:05:34.003466Z","title":"Redsearcher: A scalable and cost-efficient framework for long-horizon search agents","venue":null,"work_id":"fbfb7693-53da-4498-8b80-3e1c3d3dd9b1","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:90611f385121f5adf25b6d1fec021cf48c743780194b57794f0a2cef9841c35b","observation_id":"fbe351f8-dc2b-4f91-8b4d-b0e3a4584bae","resolution":{"observed_at":"2026-05-13T05:07:17.587389Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Harder is better: Boosting mathematical reasoning via difficulty-aware GRPO and multi-aspect question reformulation","venue":null,"work_id":"9a02ba42-e3b3-4e6a-a3d6-302a3d1440c8","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:fa855ce22d9a0996378676c79cb1ea58ce491b98b1390bc5acb7bbe56026b1cc","observation_id":"b3e77c83-4a51-4118-9c6d-fc14d1fed60d","resolution":{"observed_at":"2026-05-13T11:07:39.776771Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T16:36:22.361197Z","title":"Mind2web: Towards a generalist agent for the web.Advances in Neural Information Processing Systems, 36:28091–28114","venue":null,"work_id":"11ebf5e9-ce0a-48b6-8ddf-e267c900004c","year":2023},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:1433b66dba4789e5ba20ceadcda1fe411645916f069b0797fa93a8f652c12e74","observation_id":"de9932e6-8f5c-4a75-86a9-c4c8d28d60d8","resolution":{"observed_at":"2026-05-13T11:07:39.818141Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.17352","last_updated":"2025-11-11T08:13:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-21T17:52:43Z","title":"OpenVLThinker: Complex Vision-Language Reasoning via Iterative SFT-RL Cycles","version":3},"cited_work":{"arxiv_id":"2503.17352","doi":"10.48550/arxiv.2503.17352","metadata_source":"pith","pith_arxiv_id":"2503.17352","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OpenVLThinker: Complex Vision-Language Reasoning via Iterative SFT-RL Cycles","venue":"cs.CV","work_id":"de4c64e1-82b1-4f70-8311-a3539e7bf400","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2503.17352","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:a8b303be4f1f323c4f895092148fc1c0c7e1a402c63d938bf8cb1ab91570aa22","observation_id":"f5550734-41ba-41e7-80b5-8341adc77647","resolution":{"observed_at":"2026-05-19T06:59:03.519833Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Wildclawbench","venue":null,"work_id":"367116cb-6af4-4059-ac0a-34cfb7ac6999","year":null},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:3fce1cd52a402f85d13780428dac69b14f4000e89b66fd09fede70b1e812b3a9","observation_id":"a0d8fdd6-9f20-4899-a0f7-7fd7a4f750e6","resolution":{"observed_at":"2026-05-13T11:07:39.823623Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.16410","last_updated":"2025-05-22T09:00:19Z","snapshot_observed_at":"2026-07-06T21:28:24.644692Z","submitted_at":"2025-05-22T09:00:19Z","title":"Tool-Star: Empowering LLM-Brained Multi-Tool Reasoner via Reinforcement Learning","version":1},"cited_work":{"arxiv_id":"2505.16410","doi":"10.48550/arxiv.2505.16410","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.16410","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Tool-star: Empowering LLM-brained multi-tool reasoner via reinforcement learning","venue":"ArXiv.org","work_id":"4b5aa09c-2d1c-4591-888d-3e9aa8a1a0dc","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2505.16410","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:d663151c9c7c5ce8dbd9169c3de3222edd84f4ccb7d4984c20d38680f74a9134","observation_id":"751d4fcb-e620-4ec5-abcc-dc6947058c96","resolution":{"observed_at":"2026-05-13T05:07:17.727198Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.19849","last_updated":"2025-07-26T07:53:11Z","snapshot_observed_at":"2026-07-06T22:03:15.296567Z","submitted_at":"2025-07-26T07:53:11Z","title":"Agentic Reinforced Policy Optimization","version":1},"cited_work":{"arxiv_id":"2507.19849","doi":"10.48550/arxiv.2507.19849","metadata_source":"pith","pith_arxiv_id":"2507.19849","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Agentic Reinforced Policy Optimization","venue":"cs.LG","work_id":"6cb0d241-5e4d-44ed-9476-659819ec0681","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2507.19849","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:e2d0917338dd200d3638a5f714e8227c5cebf252dd0d7c6c7907fc2aed4e2f31","observation_id":"37eda508-fa87-42b7-be1d-f85b7e9c116e","resolution":{"observed_at":"2026-05-17T02:57:12.173630Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.18292","last_updated":"2026-04-20T14:01:10Z","snapshot_observed_at":"2026-07-06T23:05:13.178333Z","submitted_at":"2026-04-20T14:01:10Z","title":"Agent-World: Scaling Real-World Environment Synthesis for Evolving General Agent Intelligence","version":1},"cited_work":{"arxiv_id":"2604.18292","doi":"10.48550/arxiv.2604.18292","metadata_source":"pith","pith_arxiv_id":"2604.18292","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"Agent-World: Scaling Real-World Environment Synthesis for Evolving General Agent Intelligence","venue":"cs.AI","work_id":"ffadde38-12ae-49ed-b435-0d399a0ea2fd","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2604.18292","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:9de811cdafa30024fb42977107c95ae8e2fee70c0670e40ccf9bcfe7d8e1bb18","observation_id":"bec250b6-7d34-4c30-bf01-d732a8d11b3a","resolution":{"observed_at":"2026-05-13T05:07:17.621909Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09572","last_updated":"2025-04-22T17:56:22Z","snapshot_observed_at":"2026-08-04T02:37:26.383513Z","submitted_at":"2025-03-12T17:40:52Z","title":"Plan-and-Act: Improving Planning of Agents for Long-Horizon Tasks","version":3},"cited_work":{"arxiv_id":"2503.09572","doi":"10.48550/arxiv.2503.09572","metadata_source":"pith","pith_arxiv_id":"2503.09572","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Plan-and-Act: Improving Planning of Agents for Long-Horizon Tasks","venue":"cs.CL","work_id":"ee84a785-1e36-4af7-b0e8-13fc86cba1ea","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2503.09572","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:a2459c6e9579b0ad7a0db7438741e28125d31a8399881711ac0854bdf83b273a","observation_id":"5efedcd7-945a-4fa4-8eda-6e88d8374c97","resolution":{"observed_at":"2026-05-17T21:32:18.834213Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.10978","last_updated":"2025-10-28T15:11:36Z","snapshot_observed_at":"2026-07-29T19:20:21.974239Z","submitted_at":"2025-05-16T08:26:59Z","title":"Group-in-Group Policy Optimization for LLM Agent Training","version":3},"cited_work":{"arxiv_id":"2505.10978","doi":"10.48550/arxiv.2505.10978","metadata_source":"pith","pith_arxiv_id":"2505.10978","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Group-in-Group Policy Optimization for LLM Agent Training","venue":"cs.LG","work_id":"bc65d492-e6ba-4522-874c-43d2f4fc5191","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2505.10978","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:7053bf499ade799889e606a2306ec7ac17a373fe2a2d357f6b5bec8c6af84f51","observation_id":"4a61f7ee-90f2-460f-9b23-03f66a981999","resolution":{"observed_at":"2026-05-13T05:07:17.632913Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.19767","last_updated":"2025-06-24T16:31:37Z","snapshot_observed_at":"2026-08-06T23:00:22.849455Z","submitted_at":"2025-06-24T16:31:37Z","title":"SRFT: A Single-Stage Method with Supervised and Reinforcement Fine-Tuning for Reasoning","version":1},"cited_work":{"arxiv_id":"2506.19767","doi":"10.48550/arxiv.2506.19767","metadata_source":"arxiv_reference","pith_arxiv_id":"2506.19767","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2506.19767 , year=","venue":"ArXiv.org","work_id":"03d392b0-d6dc-44a8-bfdd-888ccbc9e68e","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2506.19767","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:d67d354c22b8914086ef698614158cc03dfddc6cc5307021c3d068edc6cc88ac","observation_id":"859ffaa4-7137-4c9d-bc2d-2b543d12052e","resolution":{"observed_at":"2026-05-13T05:07:17.645623Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2508.07976","doi":"10.48550/arxiv.2508.07976","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Beyond ten turns: Unlocking long-horizon agentic search with large-scale asynchronous rl","venue":"ArXiv.org","work_id":"cf9b5d87-269e-421e-aec4-52f7899c89f0","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:a5d55a4e7679a805b9e60465c0793bd25b20e2dd865b5906ae8a3e5f4ee93abe","observation_id":"1e817b8d-464e-40b4-b648-316d0bf06b9a","resolution":{"observed_at":"2026-05-13T05:07:17.649435Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.20532","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Actor-curator: Co-adaptive curriculum learning via policy-improvement bandits for rl post-training.arXiv preprint arXiv:2602.20532","venue":null,"work_id":"a8bc50a7-b768-4a64-a334-1411302abeb2","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:b307705e326ba6efcd9df176b68634496b95414ea391d0cf5b96f7ef69ef6674","observation_id":"009f5b83-d135-430d-a75b-484f7a33c050","resolution":{"observed_at":"2026-05-13T05:07:17.590870Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Deep q-learning from demonstrations","venue":null,"work_id":"58a37b8d-dae6-41eb-87bb-f614f0a8706d","year":2018},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:6ed0aa61bdb451ffa9d8fd95f41417e7faee841292fd83feea5447b0e43a00f5","observation_id":"3af81f34-30cf-47d1-bbd8-d1e391518386","resolution":{"observed_at":"2026-05-13T11:07:39.778716Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Boosting mllm reasoning with text-debiased hint-grpo","venue":null,"work_id":"fb653f7a-bc25-4aad-969d-efc8dbdea9e7","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:b219f47d81908d044810bff7bf224c0ccd379288a930596d191fee25477d9d22","observation_id":"50fbd06d-8e5a-424c-bb98-d95aa186f427","resolution":{"observed_at":"2026-05-13T11:07:39.798467Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.20802","last_updated":"2026-02-16T14:49:34Z","snapshot_observed_at":"2026-07-29T19:52:32.104228Z","submitted_at":"2026-01-28T17:45:12Z","title":"Reinforcement Learning via Self-Distillation","version":2},"cited_work":{"arxiv_id":"2601.20802","doi":"10.48550/arxiv.2601.20802","metadata_source":"pith","pith_arxiv_id":"2601.20802","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement Learning via Self-Distillation","venue":"cs.LG","work_id":"b193541d-5853-4ea4-8e4b-8e4c08617eb6","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2601.20802","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:dbccfb6865acffd8dee5e05330729f00bfffa072cc6704d97c8f69b40cce951d","observation_id":"b82336af-dcd4-47ec-bc91-ffc20befc04b","resolution":{"observed_at":"2026-05-13T05:07:17.597502Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-21T06:23:12.649775+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-21T06:23:12.649775+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Tree search for LLM agent reinforcement learning","venue":null,"work_id":"a6cdb788-4cab-4552-972f-25ecc0d4d0f0","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:20d61c03f13c43b22446c434ef208d45fdd3735d680a22cb547b5716b75bd120","observation_id":"08144029-b394-4e13-8a6d-176e06855d6c","resolution":{"observed_at":"2026-05-13T11:07:39.780576Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Thinking with map: Reinforced parallel map-augmented agent for geolocalization.ACL","venue":null,"work_id":"66f524d3-3245-4876-88c5-99db2e3be337","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:493dd69d5ad18b43fe77d48e60720dfb76a81afb8fb580fd834e79317d73d1b2","observation_id":"0be9105e-bea4-4158-beac-d4f44700b3c5","resolution":{"observed_at":"2026-05-13T11:07:39.812409Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2509.19803","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T18:30:01.575162Z","title":"Vcrl: Variance-based curriculum reinforcement learning for large language models","venue":null,"work_id":"87da4de6-b94c-43ee-8a28-bd43363bea62","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:0c6786d6f5168d6ee4be4de2887bde1a6d5808acdb94afdad3f188257c7d8037","observation_id":"8635f681-43ae-46dd-94c9-dfaef4cb5faa","resolution":{"observed_at":"2026-05-13T05:07:17.676707Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06770","last_updated":"2024-11-11T23:05:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T16:47:29Z","title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","version":3},"cited_work":{"arxiv_id":"2310.06770","doi":"10.1145/512927.512945","metadata_source":"pith","pith_arxiv_id":"2310.06770","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","venue":"cs.CL","work_id":"d0effe15-a689-441a-8e3f-ea35f1c4e4b1","year":2023},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2310.06770","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:5771eea9f03e6a91e818e2fba60da9673c68665983303c806df95222242a2d56","observation_id":"80061224-480d-4b83-b19b-6846e08e04c9","resolution":{"observed_at":"2026-05-13T05:07:17.688917Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09516","last_updated":"2025-08-05T19:08:38Z","snapshot_observed_at":"2026-07-06T20:51:28.022519Z","submitted_at":"2025-03-12T16:26:39Z","title":"Search-R1: Training LLMs to Reason and Leverage Search Engines with Reinforcement Learning","version":5},"cited_work":{"arxiv_id":"2503.09516","doi":"10.48550/arxiv.2503.09516","metadata_source":"pith","pith_arxiv_id":"2503.09516","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Search-R1: Training LLMs to Reason and Leverage Search Engines with Reinforcement Learning","venue":"cs.CL","work_id":"0e0b7549-2bc4-4574-aa7f-588ffa16eaae","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2503.09516","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:a9ac906309286c3140aebe412bcdd6724069ab42d1dcaf0635f7084f635da289","observation_id":"afda1e21-8088-47ae-9ae2-b7a6effdc9d4","resolution":{"observed_at":"2026-05-13T05:07:17.606250Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02592","last_updated":"2025-07-03T12:59:07Z","snapshot_observed_at":"2026-08-06T06:12:37.171726Z","submitted_at":"2025-07-03T12:59:07Z","title":"WebSailor: Navigating Super-human Reasoning for Web Agent","version":1},"cited_work":{"arxiv_id":"2507.02592","doi":"10.48550/arxiv.2507.02592","metadata_source":"pith","pith_arxiv_id":"2507.02592","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"WebSailor: Navigating Super-human Reasoning for Web Agent","venue":"cs.CL","work_id":"fec5a195-1dd5-425a-b552-109af948a7dd","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2507.02592","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:1033647dc7d07027cab7f4c0978778a0a78a068e6924beb4b65251b63b760818","observation_id":"52a638ee-6fb7-4688-97ab-09df64d5fd0f","resolution":{"observed_at":"2026-05-17T15:37:09.773663Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Adacurl: Adaptive curriculum reinforcement learning with invalid sample mitigation and historical revisiting","venue":null,"work_id":"7da0c753-7974-4e7c-8723-e7d29f37508c","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:2852c5618525b9874ee92b9a038eac0f629309f3cded0cd5c41e4f1cdee5ff7c","observation_id":"b970073b-9c08-4745-9f20-11aae7c9e7b9","resolution":{"observed_at":"2026-05-13T11:07:39.814168Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.21776","last_updated":"2025-10-13T12:40:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-30T16:25:25Z","title":"WebThinker: Empowering Large Reasoning Models with Deep Research Capability","version":2},"cited_work":{"arxiv_id":"2504.21776","doi":"10.48550/arxiv.2504.21776","metadata_source":"pith","pith_arxiv_id":"2504.21776","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"WebThinker: Empowering Large Reasoning Models with Deep Research Capability","venue":"cs.CL","work_id":"7e319d34-eb88-4ea9-8c0d-b66320599f98","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2504.21776","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:21fbee428ef971e9b736bf786f6ec2c24f274a836458e19592a804eaee41473e","observation_id":"55358957-318a-4682-941b-47aa643d0dac","resolution":{"observed_at":"2026-05-16T19:14:25.573630Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06892","last_updated":"2025-07-11T10:32:34Z","snapshot_observed_at":"2026-08-06T18:49:43.181677Z","submitted_at":"2025-07-09T14:29:45Z","title":"Squeeze the Soaked Sponge: Efficient Off-policy Reinforcement Finetuning for Large Language Model","version":3},"cited_work":{"arxiv_id":"2507.06892","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.06892","snapshot_observed_at":"2026-07-04T08:09:40.703268Z","title":"Squeeze the soaked sponge: Efficient off-policy reinforcement finetuning for large language model","venue":null,"work_id":"3acc1af6-eff2-4ff7-9eea-ab00848acbb3","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2507.06892","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:57ecb8d0237b0f19f3ebb74b8a9b1b104c87c13f619e2b03d6ebd9a62a7da739","observation_id":"92c5b202-9bfe-4c5c-b64a-08617c072a60","resolution":{"observed_at":"2026-05-13T05:07:17.618672Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Guided exploration with proximal policy optimization using a single demonstration","venue":null,"work_id":"e347e5ff-6f83-4823-b329-6efb3c5c3bc0","year":2021},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:5b6e75a12ad3d2e1fbbd8666d1bb39ab95f5a7e18607aab85d0b6400855c104a","observation_id":"cfd20f2a-199e-4d14-ae46-5d9da119b7bb","resolution":{"observed_at":"2026-05-13T11:07:39.821791Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T03:54:30.352907Z","title":"Truthfulqa: Measuring how models mimic hu- man falsehoods","venue":null,"work_id":"81229962-3656-44da-a54c-197faeefdb7f","year":2022},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:b9f8a3dbf99f882e4af38a96e070890a46f269787293c7c74011d86378529198","observation_id":"66b3e568-7ed0-4777-bb46-04d01342c9e7","resolution":{"observed_at":"2026-05-13T11:07:39.820085Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":"2412.19437","doi":"10.1016/j.neucom.2023.127063.url:https://www.sciencedirect","metadata_source":"pith","pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"DeepSeek-V3 Technical Report","venue":"cs.CL","work_id":"57d2791d-2219-4c31-a077-afc04b12a75c","year":2024},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:1ca40c5dbb3f0fea389f2a3abc488acce1982079267de1b8b40a6bbcb9124448","observation_id":"8caa542f-b769-4914-8565-be8a51ce2d1a","resolution":{"observed_at":"2026-05-13T05:07:17.796733Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.21460","last_updated":"2025-03-27T12:50:17Z","snapshot_observed_at":"2026-07-06T20:59:35.694800Z","submitted_at":"2025-03-27T12:50:17Z","title":"Large Language Model Agent: A Survey on Methodology, Applications and Challenges","version":1},"cited_work":{"arxiv_id":"2503.21460","doi":"10.1145/3573051.3596191","metadata_source":"pith","pith_arxiv_id":"2503.21460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Model Agent: A Survey on Methodology, Applications and Challenges","venue":"cs.CL","work_id":"4aff48d9-c46d-41d9-904b-52251e559596","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2503.21460","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:41d2d858fee013c73c93cff9edf96b38bf880cadafa8142730e664e724128718","observation_id":"2361523b-ecd7-42c7-a7aa-dac6d962aa89","resolution":{"observed_at":"2026-05-13T05:07:17.594178Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2506.07527","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T03:25:57.811550Z","title":"Learning what reinforcement learning can't: Interleaved online fine-tuning for hardest questions","venue":null,"work_id":"77400081-d7b1-4be9-88ff-c1082c417c04","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:e776ade6da911803a14cf4b230feaf911bdd3855bc0a469fbc134975e5688ef3","observation_id":"b737ab17-d520-460c-a169-a0d582078b9c","resolution":{"observed_at":"2026-05-13T05:07:17.787952Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.08377","last_updated":"2026-04-09T15:38:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-09T15:38:27Z","title":"SkillClaw: Let Skills Evolve Collectively with Agentic Evolver","version":1},"cited_work":{"arxiv_id":"2604.08377","doi":"10.48550/arxiv.2604.08377","metadata_source":"pith","pith_arxiv_id":"2604.08377","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"SkillClaw: Let Skills Evolve Collectively with Agentic Evolver","venue":"cs.AI","work_id":"31a44ebb-9aae-4719-8d62-50f8c82ece16","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2604.08377","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:bcac670fb6610b14e90a8507b003f2d9c7b435ae7c53579519b5187c8ba78ee3","observation_id":"826ac4a2-1c2d-4ade-9115-ee3388b75676","resolution":{"observed_at":"2026-05-13T05:07:17.776747Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.12983","last_updated":"2023-11-21T20:34:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-21T20:34:47Z","title":"GAIA: a benchmark for General AI Assistants","version":1},"cited_work":{"arxiv_id":"2311.12983","doi":"10.48550/arxiv.2311.12983","metadata_source":"pith","pith_arxiv_id":"2311.12983","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GAIA: a benchmark for General AI Assistants","venue":"cs.CL","work_id":"cf222b33-f7a3-4044-a570-ecfe25edb3f8","year":2023},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2311.12983","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:1f6c2133c44a2566c0bcae03a474ee360a3f19a2aaf85e123bb64ded75395557","observation_id":"cf36beac-9f5d-45f3-9e96-e84907002074","resolution":{"observed_at":"2026-05-13T05:07:17.603371Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-20T18:52:17.222295+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T18:52:17.222295+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Minimax m2.1 system card","venue":null,"work_id":"730e5b8d-7000-488c-9372-760f82058769","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:243d731689b0fc7c59886cb81c304e2bace037484c6a19e295c4be812fcd51d9","observation_id":"14e1b130-4e90-4896-aea3-d3393945bed6","resolution":{"observed_at":"2026-05-13T11:07:39.829342Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Over- coming exploration in reinforcement learning with demonstrations","venue":null,"work_id":"e26d58db-8f2b-4b6c-8b96-41183f4b6a4c","year":2018},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:c21b264b96f9c03b060ea1c16991681ab49b8fbb61163cced2d5edc9d0880920","observation_id":"bdfa9d7a-d43d-4f0b-b3dc-39f5de834860","resolution":{"observed_at":"2026-05-13T11:07:39.806867Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13923","last_updated":"2025-06-20T00:51:15Z","snapshot_observed_at":"2026-08-07T00:23:33.491596Z","submitted_at":"2025-06-16T19:03:06Z","title":"Adaptive Guidance Accelerates Reinforcement Learning of Reasoning Models","version":2},"cited_work":{"arxiv_id":"2506.13923","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.13923","snapshot_observed_at":"2026-07-03T20:48:56.210532Z","title":"Shubham Parashar, Shurui Gui, Xiner Li, Hongyi Ling, Sushil Vemuri, Blake Olson, Eric Li, Yu Zhang, James Caverlee, Dileep Kalathil, and Shuiwang Ji","venue":null,"work_id":"a0eba3eb-bd91-4d3d-bb05-c395ae033c2d","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2506.13923","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:a602d3482cbdcdd8212527f79f5db8f72a5638571946321ff7db8f0105593cd2","observation_id":"2cf22fff-26b7-4055-817f-97f6907687bc","resolution":{"observed_at":"2026-05-13T05:07:17.751601Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gpt-5.4 thinking system card","venue":null,"work_id":"f4cbca14-81cb-4c76-8915-d299cc615ba7","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:cf519cc0e572392cf8e7b851f6425782387ab93bb457e10f25ff5a553854a2a6","observation_id":"a00052b6-468d-4553-a72a-21ff3ba94931","resolution":{"observed_at":"2026-05-13T11:07:39.810613Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Iterative reasoning preference optimization.Advances in Neural Information Processing Systems, 37:116617–116637","venue":null,"work_id":"170354cc-06ed-403b-995e-f5c2c734d76a","year":2024},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:eb364dc450707a2c69d26073927d613e12a76afc111ad3ae15a2c8f777bc8d2a","observation_id":"b097e1fa-7b21-481a-acf6-bde3b98e240a","resolution":{"observed_at":"2026-05-13T11:07:39.802870Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12326","last_updated":"2025-01-21T17:48:10Z","snapshot_observed_at":"2026-07-06T20:23:58.426780Z","submitted_at":"2025-01-21T17:48:10Z","title":"UI-TARS: Pioneering Automated GUI Interaction with Native Agents","version":1},"cited_work":{"arxiv_id":"2501.12326","doi":"10.48550/arxiv.2501.12326","metadata_source":"pith","pith_arxiv_id":"2501.12326","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"UI-TARS: Pioneering Automated GUI Interaction with Native Agents","venue":"cs.AI","work_id":"0bbcf263-a46d-4525-a438-11fce3316568","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2501.12326","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:f9695b69d27788b2c0ab09bfbe55e0b26aa8f8b7cc444e894ad2300ee83141be","observation_id":"2fba3040-3165-4db1-8b1a-54dc1ca17d4e","resolution":{"observed_at":"2026-05-13T05:07:17.699644Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-19T22:22:19.638951+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-19T22:22:19.638951+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1709.10087","last_updated":"2018-06-26T13:31:37Z","snapshot_observed_at":"2026-07-06T06:01:51.192687Z","submitted_at":"2017-09-28T17:51:13Z","title":"Learning Complex Dexterous Manipulation with Deep Reinforcement Learning and Demonstrations","version":2},"cited_work":{"arxiv_id":"1709.10087","doi":"10.48550/arxiv.1709.10087","metadata_source":"pith","pith_arxiv_id":"1709.10087","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Learning Complex Dexterous Manipulation with Deep Reinforcement Learning and Demonstrations","venue":"cs.LG","work_id":"e0799bae-989e-4c06-891d-c93b0b32024d","year":2017},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/1709.10087","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:71bca165c92e018b8d012529d171926683629581ac61e6e31642861289ef964e","observation_id":"a57effd7-b998-4217-8826-62f5937f4fe7","resolution":{"observed_at":"2026-05-13T05:07:17.755910Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.12022","last_updated":"2023-11-20T18:57:34Z","snapshot_observed_at":"2026-08-04T22:55:15.345443Z","submitted_at":"2023-11-20T18:57:34Z","title":"GPQA: A Graduate-Level Google-Proof Q&A Benchmark","version":1},"cited_work":{"arxiv_id":"2311.12022","doi":"10.48550/arxiv.2311.12022","metadata_source":"pith","pith_arxiv_id":"2311.12022","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPQA: A Graduate-Level Google-Proof Q&A Benchmark","venue":"cs.AI","work_id":"9e2a976b-f5ad-4aee-af5c-243fe0fe75d2","year":2023},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2311.12022","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:b9d5bdd03b00f1a49be58851470aeb2c57d971a39c238998b54a7db87d01dcbe","observation_id":"9c44e73f-ff6c-4474-9312-9bd559a419c2","resolution":{"observed_at":"2026-05-13T05:07:17.773989Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-01T04:38:46.941438+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T04:38:46.941438+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":"1707.06347","doi":"10.1016/j.artint.2010.12.005","metadata_source":"pith","pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Proximal Policy Optimization Algorithms","venue":"cs.LG","work_id":"240c67fe-d14d-4520-91c1-38a4e272ca19","year":2017},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:9bb20358bb0a9cff775c259cb797819fdeff19e213ff1b270fbb5cfe58648c62","observation_id":"d467159e-4082-404f-8d31-199b819351da","resolution":{"observed_at":"2026-05-13T05:07:17.600157Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":"2402.03300","doi":"10.1016/0004-3702(73)90011-8","metadata_source":"pith","pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","venue":"cs.CL","work_id":"c5006563-f3ec-438a-9e35-b7b484f34828","year":2024},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:599e317725f3eac2e2e9966dd8b4f969581bc0289049e5f4a1b342ca02c2aa5f","observation_id":"1633b736-430c-4d9c-862d-6bc615c3575a","resolution":{"observed_at":"2026-05-13T05:07:17.696982Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.19897","last_updated":"2026-01-27T18:59:08Z","snapshot_observed_at":"2026-07-06T22:43:12.468415Z","submitted_at":"2026-01-27T18:59:08Z","title":"Self-Distillation Enables Continual Learning","version":1},"cited_work":{"arxiv_id":"2601.19897","doi":"10.48550/arxiv.2601.19897","metadata_source":"pith","pith_arxiv_id":"2601.19897","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-Distillation Enables Continual Learning","venue":"cs.LG","work_id":"e9aa25e3-870c-46c8-8270-e4e5948d09f0","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2601.19897","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:338f878588a66cdf3fd9d98a2ef779df0bb1dccd65449bd544eac571ec42650b","observation_id":"fc9c0f91-b266-4d95-b705-4562cdd3e576","resolution":{"observed_at":"2026-05-13T05:07:17.685825Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-23T10:52:50.091537+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T10:52:50.091537+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.03267","last_updated":"2026-05-01T23:55:43Z","snapshot_observed_at":"2026-08-02T10:52:10.211700Z","submitted_at":"2025-12-19T07:05:38Z","title":"OpenAI GPT-5 System Card","version":2},"cited_work":{"arxiv_id":"2601.03267","doi":"10.48550/arxiv.2601.03267","metadata_source":"pith","pith_arxiv_id":"2601.03267","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"OpenAI GPT-5 System Card","venue":"cs.CL","work_id":"ca87689a-0d29-4476-b504-b65dbbb08af4","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2601.03267","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:bb111d7f7b0624935e99951f1b5ae22b14a890f1705cb356a9dd7b35eca63783","observation_id":"fa15cef6-ae82-438f-9a32-97dd9aa78c15","resolution":{"observed_at":"2026-05-13T05:07:17.679643Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-03T00:38:11.458508+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T00:38:11.458508+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.02276","last_updated":"2026-02-02T16:17:38Z","snapshot_observed_at":"2026-07-06T22:44:09.804048Z","submitted_at":"2026-02-02T16:17:38Z","title":"Kimi K2.5: Visual Agentic Intelligence","version":1},"cited_work":{"arxiv_id":"2602.02276","doi":"10.48550/arxiv.2602.02276","metadata_source":"pith","pith_arxiv_id":"2602.02276","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Kimi K2.5: Visual Agentic Intelligence","venue":"cs.CL","work_id":"d690be8f-5d53-49b0-b1e7-79668eb8fcdb","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2602.02276","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:9a5abb88989274d4c57621cbe248abd721a5b56e4c0ce94efefc06582d6ee72e","observation_id":"95c47958-1b89-4fc5-ad6c-b886c217c52a","resolution":{"observed_at":"2026-05-13T05:07:17.768488Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T10:04:52.365581Z","title":"Qwen3.5: Accelerating productivity with native multimodal agents, February","venue":null,"work_id":"f37dcb7f-3329-4b47-b2b7-85458b1592d1","year":null},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:ddfe87eb2ae93259acb247e0f6586c963319a7efb4e1e976d53c8290c3f3cf84","observation_id":"485864f6-187f-404a-a5ab-3f6c1c9529ab","resolution":{"observed_at":"2026-05-13T11:07:39.805037Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T10:04:52.368885Z","title":null,"venue":null,"work_id":"c39e4053-229e-4a4c-8726-7c6ff6b49736","year":null},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:291055aa3cdcebbf9df0fea32ec513fb48662278039a83639836b53a47c8b409","observation_id":"e2912aa8-4b67-4e68-b97c-f70d1e74a105","resolution":{"observed_at":"2026-05-13T11:07:39.800591Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.24701","last_updated":"2026-05-18T04:10:32Z","snapshot_observed_at":"2026-07-06T22:34:18.297603Z","submitted_at":"2025-10-28T17:53:02Z","title":"Tongyi DeepResearch Technical Report","version":3},"cited_work":{"arxiv_id":"2510.24701","doi":"10.48550/arxiv.2510.24701","metadata_source":"pith","pith_arxiv_id":"2510.24701","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Tongyi DeepResearch Technical Report","venue":"cs.CL","work_id":"1c8db01b-b50f-4711-b181-a04bcc3e9aa8","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2510.24701","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:d09ab1dd5aaf353f5a6948080dbaa3b591381e18451e4d898f50b15a0d4b08cf","observation_id":"68746824-b279-4f01-862c-ff4041c7e0d0","resolution":{"observed_at":"2026-05-15T08:56:57.433092Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.52202/075280-3275","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Language models don’t always say what they think: Unfaithful explanations in chain-of-thought prompting.Advances in Neural Information Processing Systems, 36:74952–74965","venue":"Advances in Neural Information Processing Systems 36","work_id":"20491003-4ee3-4877-8191-ba12773d0f0b","year":2023},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:0975782c3867ce2f560c5a3431867ffffc07808b31786188285ddb4f6d0492b1","observation_id":"f4e2c5f8-93f7-4bc0-b219-a26eaf80f776","resolution":{"observed_at":"2026-05-13T11:07:39.774687Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:14.455455+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:14.455455+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1812.02648","last_updated":"2018-12-06T16:36:20Z","snapshot_observed_at":"2026-07-06T07:19:38.028327Z","submitted_at":"2018-12-06T16:36:20Z","title":"Deep Reinforcement Learning and the Deadly Triad","version":1},"cited_work":{"arxiv_id":"1812.02648","doi":"10.48550/arxiv.1812.02648","metadata_source":"pith","pith_arxiv_id":"1812.02648","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Deep Reinforcement Learning and the Deadly Triad","venue":"cs.AI","work_id":"de214ead-4cb0-4abd-be3d-ae3389f55e9b","year":2018},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/1812.02648","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:acabcd2b4368d4a923577558e0a3713617adbd8d3608e34af024613a5a2f0d93","observation_id":"8dc39b33-8e3c-4204-ba86-c86d6618adae","resolution":{"observed_at":"2026-05-13T05:07:17.711777Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1707.08817","last_updated":"2018-10-08T13:38:52Z","snapshot_observed_at":"2026-08-02T04:07:03.272158Z","submitted_at":"2017-07-27T11:16:53Z","title":"Leveraging Demonstrations for Deep Reinforcement Learning on Robotics Problems with Sparse Rewards","version":2},"cited_work":{"arxiv_id":"1707.08817","doi":null,"metadata_source":"pith","pith_arxiv_id":"1707.08817","snapshot_observed_at":"2026-07-10T21:47:36.312721Z","title":"Leveraging Demonstrations for Deep Reinforcement Learning on Robotics Problems with Sparse Rewards","venue":"cs.AI","work_id":"a4478529-fa6c-4c19-98bd-743c55919fd6","year":2017},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/1707.08817","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:2163e400baec0a6b503cc2427194bd4bee8732659d1261a3cb62a3400d56eaa6","observation_id":"87baeeb7-af4a-4cee-b02e-f8467ffb4008","resolution":{"observed_at":"2026-05-13T05:07:17.758867Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.24873","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-01T21:26:14.207661Z","title":"Let it flow: Agentic crafting on rock and roll, building the rome model within an open agentic learning ecosystem","venue":null,"work_id":"a1afde43-96e3-49c9-af14-25f128d65fe3","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:9734e914b7625be9b15f4ab1cb1c225e49332996e92a5cb44ce00181eaf2734f","observation_id":"74e5c648-4a6d-4977-9805-bacabdda8f9f","resolution":{"observed_at":"2026-05-13T05:07:17.761647Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.10165","last_updated":"2026-05-11T10:03:41Z","snapshot_observed_at":"2026-07-06T22:48:35.345421Z","submitted_at":"2026-03-10T18:59:01Z","title":"OpenClaw-RL: Train Any Agent Simply by Talking","version":2},"cited_work":{"arxiv_id":"2603.10165","doi":"10.48550/arxiv.2603.10165","metadata_source":"pith","pith_arxiv_id":"2603.10165","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OpenClaw-RL: Train Any Agent Simply by Talking","venue":"cs.CL","work_id":"78607317-8305-4515-8dc3-20b4ff5b8f3a","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2603.10165","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:e78ece9fc2697d9e6e784c11750c9d8536322c4418ad6cdea99918aaff1f2494","observation_id":"db34b9fc-5d71-432d-993e-62773e8d8907","resolution":{"observed_at":"2026-05-13T05:07:17.765019Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:b04cb0b98f20f017753fa5ed8196245fe3d2d0a43660e650c016da690a957d96","observation_id":"7711b20f-5074-4cc0-b9e5-82086d47af10","resolution":{"observed_at":"2026-05-13T07:13:34.687091Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.12538","last_updated":"2026-01-18T18:58:23Z","snapshot_observed_at":"2026-08-04T22:42:23.171653Z","submitted_at":"2026-01-18T18:58:23Z","title":"Agentic Reasoning for Large Language Models","version":1},"cited_work":{"arxiv_id":"2601.12538","doi":"10.48550/arxiv.2601.12538","metadata_source":"pith","pith_arxiv_id":"2601.12538","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Agentic Reasoning for Large Language Models","venue":"cs.AI","work_id":"062546cd-e1a7-46e4-b617-c6b8e19b6fa3","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2601.12538","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:0e67a75ce34b19ed44741352f68fa00d5a769b0d9b4774ce4378474153c76eec","observation_id":"f9d25609-d38b-42b2-8332-d678f8004c54","resolution":{"observed_at":"2026-05-17T15:14:26.657956Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-01T10:08:10.217033+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T10:08:10.217033+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.07572","last_updated":"2025-08-10T05:59:20Z","snapshot_observed_at":"2026-08-04T14:46:05.418102Z","submitted_at":"2025-01-13T18:58:07Z","title":"WebWalker: Benchmarking LLMs in Web Traversal","version":3},"cited_work":{"arxiv_id":"2501.07572","doi":"10.48550/arxiv.2501.07572","metadata_source":"pith","pith_arxiv_id":"2501.07572","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Jialong Wu, Baixuan Li, Runnan Fang, Wenbiao Yin, Liwen Zhang, Zhengwei Tao, Dingchu Zhang, Zekun Xi, Gang Fu, Yong Jiang, Pengjun Xie, Fei Huang, and Jingren Zhou","venue":"cs.CL","work_id":"8528e4cd-bcbc-4f57-9bd2-11ce86dd3493","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2501.07572","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:a3461d4549bdc843452ba3f9834db0f652ffeba4319dbcd01cf34ab6296c70ab","observation_id":"f4c1fd2f-5f2e-44e6-af5b-fb0dce07b088","resolution":{"observed_at":"2026-05-13T05:07:17.791143Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.01223","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T16:24:01.012815Z","title":"Learn hard problems during rl with reference guided fine-tuning","venue":null,"work_id":"cca99282-a0f6-4e3b-894f-0589afedb0de","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:59d2d739c48975530dc695bc1f5ff9ec315227b17d190468f70e22da5f14d9ae","observation_id":"82846ed5-0a12-479c-a3b3-b13b3b5eddd7","resolution":{"observed_at":"2026-05-13T05:07:17.702708Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Osworld: Benchmarking multimodal agents for open-ended tasks in real computer environments","venue":null,"work_id":"8866ac21-13df-424b-a428-ab1ffff49b2e","year":2024},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:bd62a372c2b1d688b3e3dd7fa18dc7d3f3a3402ee8d5eaf00e4001ba07cf3228","observation_id":"a51275f2-6cbc-4d04-9b51-8f2c2dfe705d","resolution":{"observed_at":"2026-05-13T11:07:39.808900Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.14945","last_updated":"2025-06-22T00:18:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-21T08:09:13Z","title":"Learning to Reason under Off-Policy Guidance","version":5},"cited_work":{"arxiv_id":"2504.14945","doi":"10.48550/arxiv.2504.14945","metadata_source":"pith","pith_arxiv_id":"2504.14945","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Learning to Reason under Off-Policy Guidance","venue":"cs.LG","work_id":"4ebcdbe2-5000-4f58-a7e0-aa9ae381b684","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2504.14945","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:47cc14047483ddd40c9f60a0ddef752ad6cd297f9491b640cbd16253500708ac","observation_id":"1d69ddf5-36fb-4f9f-ae7f-c240932e4ed3","resolution":{"observed_at":"2026-05-15T23:17:03.075876Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.22190","last_updated":"2026-05-25T21:32:44Z","snapshot_observed_at":"2026-08-02T20:51:33.578240Z","submitted_at":"2026-02-25T18:34:57Z","title":"GUI-Libra: Training Native GUI Agents to Reason and Act with Action-aware Supervision and Partially Verifiable RL","version":2},"cited_work":{"arxiv_id":"2602.22190","doi":null,"metadata_source":"pith","pith_arxiv_id":"2602.22190","snapshot_observed_at":"2026-07-02T11:56:55.516310Z","title":"Gui-libra: Training native gui agents to reason and act with action-aware supervision and partially verifiable rl","venue":"cs.LG","work_id":"92549a20-b67b-4432-a2f1-81176c3e0d06","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2602.22190","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:d9dc7f2876d42abf0fd400915f6daddbf51c54c0cf280ee0463e85843be39632","observation_id":"64b3bdc3-82c7-4ed2-8889-7e699ce8bea5","resolution":{"observed_at":"2026-05-27T02:04:34.683381Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T18:37:32.087006Z","title":"React: Synergizing reasoning and acting in language models","venue":null,"work_id":"404308e4-3b7a-4845-8899-d58a0072d6ed","year":2022},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:c52b7f4cae5e184e355834573eb997eecb814f92aa31990c540f974ee25c2ea1","observation_id":"9f6a887b-9d47-4c88-ae0f-10733a8b4686","resolution":{"observed_at":"2026-05-13T11:07:39.827410Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12045","last_updated":"2024-06-17T19:33:08Z","snapshot_observed_at":"2026-08-02T22:19:29.043854Z","submitted_at":"2024-06-17T19:33:08Z","title":"$\\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","version":1},"cited_work":{"arxiv_id":"2406.12045","doi":"10.48550/arxiv.2406.12045","metadata_source":"pith","pith_arxiv_id":"2406.12045","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"$\\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","venue":"cs.AI","work_id":"6a8d8dc4-0cc0-4052-8109-abbcdcd4a962","year":2024},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2406.12045","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:d7a14b4497449771adc8d6256af51763ed1632bbb44a054eecf553c12e5dcf76","observation_id":"afad9ae9-cc3b-4580-a25f-ea603716bab3","resolution":{"observed_at":"2026-05-13T05:07:17.666389Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:21.86453+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:21.86453+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.03048","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T04:27:36.740990Z","title":"Coba-rl: Capability-oriented budget allocation for reinforcement learning in llms","venue":null,"work_id":"dfc92b4c-afc7-4195-b7d6-d28f98ea47f6","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:2092dce41bd36dcc899c9307ff6183589934c2e3531a5addf2983d2123b1735c","observation_id":"69d3835c-60f0-442b-9165-e06fad3beb15","resolution":{"observed_at":"2026-05-13T05:07:17.669790Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.21383","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2603.21383 , year=","venue":null,"work_id":"97f6da2e-0071-494f-976b-8422ffc48b69","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:528f45a59db6f6f594d95ba1f1f532d95ac1b1d7b97831e38bda4623c6c91a8d","observation_id":"7f6c0cb2-ad96-4afd-8d81-644cb60a574a","resolution":{"observed_at":"2026-05-13T05:07:17.663444Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.14880","last_updated":"2025-09-01T15:33:47Z","snapshot_observed_at":"2026-08-06T02:11:55.637213Z","submitted_at":"2025-08-20T17:51:20Z","title":"MedResearcher-R1: Expert-Level Medical Deep Researcher via A Knowledge-Informed Trajectory Synthesis Framework","version":3},"cited_work":{"arxiv_id":"2508.14880","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.14880","snapshot_observed_at":"2026-06-29T21:33:58.797000Z","title":"Medresearcher-r1: Expert-level medical deep researcher via a knowledge-informed trajectory synthesis framework","venue":null,"work_id":"adb64f70-9b5d-47c5-b54f-0a5eb958a55b","year":2018},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2508.14880","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:7385c30e1e9722200a30596e7db017006c0feadd4a5591606468f28c608f6076","observation_id":"41b76d43-58f0-4bec-a003-97980b44661c","resolution":{"observed_at":"2026-05-13T05:07:17.656657Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.14476","last_updated":"2025-05-20T01:37:34Z","snapshot_observed_at":"2026-08-02T01:40:54.187278Z","submitted_at":"2025-03-18T17:49:06Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","version":2},"cited_work":{"arxiv_id":"2503.14476","doi":"10.48550/arxiv.2503.14476","metadata_source":"pith","pith_arxiv_id":"2503.14476","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","venue":"cs.LG","work_id":"64019d00-0b11-4bbd-b173-b46c8fad0157","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2503.14476","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:84acd8926ecdb4bf55a09f1ede7ebef10722401ecad63918cedbbb600a1f69c6","observation_id":"1c777d36-c31c-4b72-bea6-5de65d9332bc","resolution":{"observed_at":"2026-05-13T05:07:17.694521Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-24T09:23:06.254602+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-24T09:23:06.254602+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.13837","last_updated":"2025-11-24T06:11:04Z","snapshot_observed_at":"2026-07-06T21:11:34.701779Z","submitted_at":"2025-04-18T17:59:56Z","title":"Does Reinforcement Learning Really Incentivize Reasoning Capacity in LLMs Beyond the Base Model?","version":5},"cited_work":{"arxiv_id":"2504.13837","doi":"10.48550/arxiv.2504.13837","metadata_source":"pith","pith_arxiv_id":"2504.13837","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Does Reinforcement Learning Really Incentivize Reasoning Capacity in LLMs Beyond the Base Model?","venue":"cs.AI","work_id":"d854765a-e664-41c0-8655-21c4bf2e0cc4","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2504.13837","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:f7063997fe32e359f51f982dc5ce04c16ea7191bbbdd268828524b30018c9dd6","observation_id":"66a6d182-aa4a-4d38-b0c8-a96694d6572c","resolution":{"observed_at":"2026-05-13T05:07:17.659881Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:42.002839+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:42.002839+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.10395","doi":"10.48550/arxiv.2511.10395","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Agentevolver: Towards efficient self-evolving agent system","venue":"ArXiv.org","work_id":"e9710c06-579f-415b-88ba-965ce465b757","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:6f50e64ad9b959a66aad1e00a103934343906c35626599ea7f8dd444774a16a8","observation_id":"a850c3b3-a2dc-4ad3-8e8c-acf481ef8173","resolution":{"observed_at":"2026-05-13T05:07:17.785140Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.02547","last_updated":"2026-04-17T18:09:08Z","snapshot_observed_at":"2026-08-03T09:07:42.489237Z","submitted_at":"2025-09-02T17:46:26Z","title":"The Landscape of Agentic Reinforcement Learning for LLMs: A Survey","version":5},"cited_work":{"arxiv_id":"2509.02547","doi":"10.48550/arxiv.2509.02547","metadata_source":"pith","pith_arxiv_id":"2509.02547","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"The Landscape of Agentic Reinforcement Learning for LLMs: A Survey","venue":"cs.AI","work_id":"87909127-da20-4ccc-8ae3-4a4a20ef81b7","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2509.02547","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:8ba5bc987f1e0c5d367c52fd2953d8ce36324ca3f404f35cb1ad876c95bd0996","observation_id":"ff787be5-ad14-4556-bcbd-6f07b151209e","resolution":{"observed_at":"2026-05-13T05:07:17.608879Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2508.11408","doi":"10.48550/arxiv.2508.11408","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2508.11408 , year=","venue":"arXiv (Cornell University)","work_id":"994df3c4-5dc6-45c0-9db5-e82e415ce5d8","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:9f5cd6bdff311b94d89eef51a2504506b17e3230ffef3d8f08b7b2096980784e","observation_id":"5eddc1e9-79b6-4ecb-998d-34d290444f99","resolution":{"observed_at":"2026-05-13T05:07:17.705737Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.18734","last_updated":"2026-03-20T15:40:19Z","snapshot_observed_at":"2026-08-06T08:49:57.295985Z","submitted_at":"2026-01-26T17:56:50Z","title":"Self-Distilled Reasoner: On-Policy Self-Distillation for Large Language Models","version":3},"cited_work":{"arxiv_id":"2601.18734","doi":"10.18653/v1/2025.emnlp-main.125.https://aclanthology.org/2025.emnlp-main.125/","metadata_source":"pith","pith_arxiv_id":"2601.18734","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Self-Distilled Reasoner: On-Policy Self-Distillation for Large Language Models","venue":"cs.LG","work_id":"bae00e84-9b0d-433d-a066-20b951f0b4d0","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2601.18734","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:d0d5ac4e688e3cf667c293afd7f46e8367d2b156df2ef0b90d5df8a5332b7e54","observation_id":"d45a8540-4c29-4f0d-8030-aef3cc060ce8","resolution":{"observed_at":"2026-05-13T05:07:17.779421Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.01161","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T22:26:37.214162Z","title":"Prosperity before collapse: How far can off-policy rl reach with stale data on llms?","venue":null,"work_id":"99a02056-55bc-45e6-8d10-7f31cf541473","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:647a6f1a05f8cd7c58caa030af9282157792b6bc578ec4fffef003af0876e219","observation_id":"4bd1baa4-5550-45eb-b3a4-96e166e18099","resolution":{"observed_at":"2026-05-13T05:07:17.708884Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.09856","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T17:09:59.313952Z","title":"Code2world: A gui world model via renderable code generation","venue":null,"work_id":"9ee28d94-edc8-4660-81eb-40b5d9e02315","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:0ca8b31cbb2528ef20e5555665f8b9642aa86a84110acddde824e4cc4bc966bb","observation_id":"379ba4ab-3a65-469a-bd7e-d9cdcdfc50cf","resolution":{"observed_at":"2026-05-13T05:07:17.639387Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07911","last_updated":"2023-11-14T05:13:55Z","snapshot_observed_at":"2026-07-06T16:47:08.877195Z","submitted_at":"2023-11-14T05:13:55Z","title":"Instruction-Following Evaluation for Large Language Models","version":1},"cited_work":{"arxiv_id":"2311.07911","doi":"10.48550/arxiv.2311.07911","metadata_source":"pith","pith_arxiv_id":"2311.07911","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Instruction-Following Evaluation for Large Language Models","venue":"cs.CL","work_id":"3aa06177-125a-4f5a-8f4a-8070c5986c26","year":2023},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2311.07911","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:08b36df43134095670d3efc53a8ccdce63609c1f05866a7fcace32c38de437ba","observation_id":"66043e16-8a57-43b6-9729-a26b0da6908c","resolution":{"observed_at":"2026-05-13T05:07:17.642066Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.19314","last_updated":"2025-05-01T05:02:57Z","snapshot_observed_at":"2026-08-06T05:37:40.638404Z","submitted_at":"2025-04-27T17:32:43Z","title":"BrowseComp-ZH: Benchmarking Web Browsing Ability of Large Language Models in Chinese","version":2},"cited_work":{"arxiv_id":"2504.19314","doi":"10.48550/arxiv.2504.19314","metadata_source":"pith","pith_arxiv_id":"2504.19314","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"BrowseComp-ZH: Benchmarking Web Browsing Ability of Large Language Models in Chinese","venue":"cs.CL","work_id":"d2997896-a54e-43f5-80ac-d4d548116d21","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2504.19314","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:e6fd392ac66cbf4baab51b3bbe2b80695b213b09eb8e20c8147f01a28a094a59","observation_id":"3bb80b65-9f3d-4e17-a552-07be46bb0b8d","resolution":{"observed_at":"2026-05-17T22:04:50.009871Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance"},"reference_resolution":{"displayed":82,"state_counts":{"malformed_identifier":0,"metadata_mismatch":4,"parse_uncertain":0,"unresolved":1,"verified_exact":56,"verified_fuzzy":21},"total_outbound_references":82},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 82 of 82 outbound references and 1 inbound Pith citation observation for arXiv:2605.12004."}