{"as_of":"2026-08-09T16:46:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:ca29a8ce118853416fa023d324c9b119a6021cfc5ef1b560c9560c8d1f9194c4","coverage":[{"denominator":39,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":39,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T18:47:23.367773Z","state":"measured"},{"denominator":39,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":39,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.07441/citation-record","integrity":"/paper/2507.07441/integrity","json":"/paper/2507.07441/citation-record.json","paper":"/paper/2507.07441"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:19.026216Z","title":"online\" 'onlinestring :=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:19.026216Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:1f38e29fbca583af60e442119da144ca37b762d76a7a59e905e596facf8fcc2a","observation_id":"d1c58432-c1bf-4e9d-9d6d-d27380de17db","resolution":{"observed_at":"2026-08-06T18:47:19.026216Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:19.120230Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:19.120230Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:252ab28d9f4e4e0ad3a878c25e21fb3194f7d886d38f496a7814f8852b6a70cd","observation_id":"b1345c0c-97ec-4bdb-ba76-79784be37139","resolution":{"observed_at":"2026-08-06T18:47:19.120230Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-06T18:47:19.247638Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:19.247638Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:e685a59b30bd14c6c3b6c523b14cdbd73e773489c988ba53304f8c5c57e73604","observation_id":"1f951891-0a63-4b3b-98d3-0325132dc1e0","resolution":{"observed_at":"2026-08-06T18:47:19.247638Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.05915","last_updated":"2023-10-09T17:58:38Z","snapshot_observed_at":"2026-08-05T18:32:49.850038Z","submitted_at":"2023-10-09T17:58:38Z","title":"FireAct: Toward Language Agent Fine-tuning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.05915","snapshot_observed_at":"2026-08-06T18:47:19.417248Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:19.417248Z"},"links":{"cited_paper":"/paper/2310.05915","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:0eb2c1fa8d34283c06791249a99a6d13e2ea0b36daa854c50d8d7b3404efabbd","observation_id":"4ece5b4b-0d92-46bc-a299-ae61de169c11","resolution":{"observed_at":"2026-08-06T18:47:19.417248Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:24.142316Z","title":null,"venue":null,"work_id":"db18f7e9-b782-41d0-83e6-0b731263929c","year":2024},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:19.555232Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:4fd42adbfc706e4c0e9a34d4ba1b5f3ff120f8ca45154759d925cadb7d579b43","observation_id":"f1b399fd-71c6-495a-a48b-3b945872a3c1","resolution":{"observed_at":"2026-08-06T18:47:24.145308Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.02197","last_updated":"2025-06-05T01:42:08Z","snapshot_observed_at":"2026-08-08T13:58:18.116445Z","submitted_at":"2025-03-04T02:14:55Z","title":"ATLaS: Agent Tuning via Learning Critical Steps","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.02197","snapshot_observed_at":"2026-08-06T18:47:19.688593Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:19.688593Z"},"links":{"cited_paper":"/paper/2503.02197","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:aa744587ab4d7c07e6ce78a5af8ea2871b42498a0380ca9646ca34b9bc43d4a9","observation_id":"febc73ab-b7d9-42e1-9fb6-6c1cc19d2ae6","resolution":{"observed_at":"2026-08-06T18:47:19.688593Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-06T18:47:19.800141Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:19.800141Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:5f691fc4fa7ca76d65f345fe85f68c7d2cfb611d7c90dd69ee3e56b839e8aae8","observation_id":"2c1495a5-144f-40fd-bcbb-48e8e818efe4","resolution":{"observed_at":"2026-08-06T18:47:19.800141Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16339","last_updated":"2025-01-08T20:11:59Z","snapshot_observed_at":"2026-08-07T23:13:03.652587Z","submitted_at":"2024-12-20T21:00:11Z","title":"Deliberative Alignment: Reasoning Enables Safer Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.16339","snapshot_observed_at":"2026-08-06T18:47:19.985383Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:19.985383Z"},"links":{"cited_paper":"/paper/2412.16339","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:886e02b51baf300166b6699b9536707cd5521fbf438351f180db7d06825b8c50","observation_id":"e922bb64-c738-4675-aed0-39776862d27a","resolution":{"observed_at":"2026-08-06T18:47:19.985383Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.11143","last_updated":"2025-10-09T12:22:46Z","snapshot_observed_at":"2026-07-31T12:28:37.704994Z","submitted_at":"2024-05-20T01:04:40Z","title":"OpenRLHF: An Easy-to-use, Scalable and High-performance RLHF Framework","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.11143","snapshot_observed_at":"2026-08-06T18:47:20.145227Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:20.145227Z"},"links":{"cited_paper":"/paper/2405.11143","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:c09fc5d06946edd03f98b543d5d669b850daaadd065346f54757a329a32129d3","observation_id":"d479af49-068f-4c8a-b185-056e521f23de","resolution":{"observed_at":"2026-08-06T18:47:20.145227Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:20.251707Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:20.251707Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:fd802190fe61fc07477bae6d45376a6dfb719d9479ad932d8b7f8047d09078b6","observation_id":"83ca68c7-130b-41d1-95ca-d456ecddac34","resolution":{"observed_at":"2026-08-06T18:47:20.251707Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.14507","last_updated":"2024-09-18T09:25:20Z","snapshot_observed_at":"2026-07-06T18:49:10.964379Z","submitted_at":"2024-07-19T17:59:03Z","title":"Internal Consistency and Self-Feedback in Large Language Models: A Survey","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.14507","snapshot_observed_at":"2026-08-06T18:47:20.378177Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:20.378177Z"},"links":{"cited_paper":"/paper/2407.14507","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:c7078b00262df4d064268956b36ddbc2dc41e064439b1abc3d70f0a94ef4b060","observation_id":"9c83b9cf-8803-4826-b029-1e45f8a4ea5d","resolution":{"observed_at":"2026-08-06T18:47:20.378177Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.02584","last_updated":"2025-02-04T18:58:31Z","snapshot_observed_at":"2026-08-09T11:38:31.414567Z","submitted_at":"2025-02-04T18:58:31Z","title":"QLASS: Boosting Language Agent Inference via Q-Guided Stepwise Search","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.02584","snapshot_observed_at":"2026-08-06T18:47:20.506045Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:20.506045Z"},"links":{"cited_paper":"/paper/2502.02584","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:e32d62fffebe95d58f27735241aaee1a0b0465cedb3aa1ad92d53cebbe67d7fc","observation_id":"bcac6f08-a539-4ce4-a1a4-233b120afa61","resolution":{"observed_at":"2026-08-06T18:47:20.506045Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:20.615254Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:20.615254Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:7f13f969770ea956a1ef9c2155f25fb40804a7b3bd1e31ba599e146b750914c2","observation_id":"dfeec77f-50a9-43e2-98f4-73c923cc7f89","resolution":{"observed_at":"2026-08-06T18:47:20.615254Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.09332","last_updated":"2022-06-01T19:08:11Z","snapshot_observed_at":"2026-08-07T17:14:39.278754Z","submitted_at":"2021-12-17T05:43:43Z","title":"WebGPT: Browser-assisted question-answering with human feedback","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.09332","snapshot_observed_at":"2026-08-06T18:47:20.674156Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:20.674156Z"},"links":{"cited_paper":"/paper/2112.09332","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:fdc0e67c55ca820177b059fc3d2aa5e14d9c14156949bc8b3bdf20201a1bb93e","observation_id":"2a1dce29-d0dc-4b0b-ad32-2a4042ff241d","resolution":{"observed_at":"2026-08-06T18:47:20.674156Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:24.120202Z","title":null,"venue":null,"work_id":"6f691064-736c-4a62-a423-653b98d5a9d9","year":2024},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:20.801364Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:68f4c313ca8f4191cb57435ecc42e0fa71600b21f99023e839d6a7cd528cca30","observation_id":"0e540441-45df-4b22-9e46-e1c6f479d971","resolution":{"observed_at":"2026-08-06T18:47:24.123063Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.emnlp-main.138","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:23.496439Z","title":null,"venue":null,"work_id":"aade76b0-2578-4e3f-867f-b1b181145384","year":2024},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:20.899057Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:0d8fa9b3ac837f92a935c03cefd2b5a4bec5b3feb74443c5ff1d73c6f047d1d5","observation_id":"b9d1d53f-8349-4011-adc4-2a88be6ebd52","resolution":{"observed_at":"2026-08-06T18:47:23.561522Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:20.991436Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:20.991436Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:025b6469d28e37cb2733c8a53da98b27f6cd7a780dd1849725088fde69dc89c8","observation_id":"e1f0d83a-7e41-47e9-bcfb-b47baaca3f27","resolution":{"observed_at":"2026-08-06T18:47:20.991436Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.03768","last_updated":"2021-03-14T22:44:38Z","snapshot_observed_at":"2026-08-08T17:40:47.037804Z","submitted_at":"2020-10-08T05:13:36Z","title":"ALFWorld: Aligning Text and Embodied Environments for Interactive Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.03768","snapshot_observed_at":"2026-08-06T18:47:21.086928Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:21.086928Z"},"links":{"cited_paper":"/paper/2010.03768","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:d4eb4ffed669637933539d944046ba964f795c1ac4afb9de07f4f8aa73b68f5f","observation_id":"e8dd96be-fad7-46d0-b042-04279887dcd1","resolution":{"observed_at":"2026-08-06T18:47:21.086928Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:24.103707Z","title":null,"venue":null,"work_id":"942c58b7-b6bb-42c6-9276-17178bb80712","year":2024},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:21.253533Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:bce1ce492a91f1779dec236e2d780333309e55bb43fbdf611c042cc5e4daa27f","observation_id":"ad2e3530-7d07-41c3-a562-a53f5142c921","resolution":{"observed_at":"2026-08-06T18:47:24.106815Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:21.432756Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:21.432756Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:a2eff0bd2308a949d4262cb9edc09678cc9cf846cf884ab4fa4adffc35ffd1bd","observation_id":"7d8b0230-658b-4639-bf03-1e40356300d7","resolution":{"observed_at":"2026-08-06T18:47:21.432756Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:24.094537Z","title":null,"venue":null,"work_id":"a7dccfa6-ad60-4fc5-a8ab-4ce6abde3f31","year":2025},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:21.561733Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:88dfe6f4d9882f01f38803f95eb3fbe1ade81af7abccb2a001c470492991da76","observation_id":"165adf27-7a16-4ede-b22a-c2a605c72513","resolution":{"observed_at":"2026-08-06T18:47:24.097308Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.07540","last_updated":"2022-11-14T17:52:27Z","snapshot_observed_at":"2026-08-06T07:25:42.562786Z","submitted_at":"2022-03-14T22:52:34Z","title":"ScienceWorld: Is your Agent Smarter than a 5th Grader?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.07540","snapshot_observed_at":"2026-08-06T18:47:21.659567Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:21.659567Z"},"links":{"cited_paper":"/paper/2203.07540","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:9f7af965588c702276432a424d3df98e04809e220407097005a8f4c5e89729f4","observation_id":"1cad929e-0d5f-4796-9417-4ee8127b40f1","resolution":{"observed_at":"2026-08-06T18:47:21.659567Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:21.809976Z","title":"Chi, Sharan Narang, Aakanksha Chowdhery, and Denny Zhou","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:21.809976Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:25bf2f309b42e16b2b4d39dc2df654a1ee343ced98671d34973834109db99a78","observation_id":"a6362ff3-168c-41a8-a62c-e50eb85720ec","resolution":{"observed_at":"2026-08-06T18:47:21.809976Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:21.904965Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:21.904965Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:a3efe5131cd9eb667489317fa3a604433458d187f011349ba73843d94b72c514","observation_id":"f015df02-002d-4ea6-832d-a0b9e3e87cd4","resolution":{"observed_at":"2026-08-06T18:47:21.904965Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.18407","last_updated":"2025-02-25T17:58:02Z","snapshot_observed_at":"2026-08-07T17:49:27.969414Z","submitted_at":"2025-02-25T17:58:02Z","title":"AgentRM: Enhancing Agent Generalization with Reward Modeling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.18407","snapshot_observed_at":"2026-08-06T18:47:22.078905Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:22.078905Z"},"links":{"cited_paper":"/paper/2502.18407","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:cf4c9bef28b9360ff9afc6e7d40c287ee7ef9391aae4d282e17af7a28bdbfec9","observation_id":"b0bc5883-362f-47e2-b5bd-e51e120a841e","resolution":{"observed_at":"2026-08-06T18:47:22.078905Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03136","last_updated":"2025-08-20T03:44:04Z","snapshot_observed_at":"2026-07-06T19:27:29.609371Z","submitted_at":"2024-10-04T04:23:36Z","title":"Deliberate Reasoning in Language Models as Structure-Aware Planning with an Accurate World Model","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.03136","snapshot_observed_at":"2026-08-06T18:47:22.180797Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:22.180797Z"},"links":{"cited_paper":"/paper/2410.03136","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:6c8e070510c96f2325bfb92339617732ace0234444c0a7599516d6542a5cf779","observation_id":"099e01cc-e3d8-45c5-af70-45b54d5a7a0c","resolution":{"observed_at":"2026-08-06T18:47:22.180797Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.02682","last_updated":"2025-09-10T16:45:42Z","snapshot_observed_at":"2026-08-07T17:30:14.136096Z","submitted_at":"2025-03-04T14:54:45Z","title":"MPO: Boosting LLM Agents with Meta Plan Optimization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.02682","snapshot_observed_at":"2026-08-06T18:47:22.268797Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:22.268797Z"},"links":{"cited_paper":"/paper/2503.02682","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:883a278c637f9d3e208376914b89cfdae6305039cb52f014e982956738fb2c3f","observation_id":"8a51431e-3524-489a-8c88-44714305f71a","resolution":{"observed_at":"2026-08-06T18:47:22.268797Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:22.338609Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:22.338609Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:f7c3deb3dc837c4d9881cb53821cac56bec3140dea4e0edcaa33e369bd10c8ac","observation_id":"192b2273-cd1a-48aa-ac22-ec79c06ddf31","resolution":{"observed_at":"2026-08-06T18:47:22.338609Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-08-06T18:47:22.420291Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:22.420291Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:0e6535ddc3de8d3cb8fe0db690791d6904244a6adeb977aa5d2b2792eb3b3d2e","observation_id":"eb61d8e1-ed56-40a8-80e1-784696faf346","resolution":{"observed_at":"2026-08-06T18:47:22.420291Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:22.629414Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:22.629414Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:550b96db2d9e4d9ccfeb3c98a13bc240cfda8fba1eba9948b81ed31da6d73f0a","observation_id":"ad3bb6f5-ff79-4a3f-b4b1-70497cacb9c7","resolution":{"observed_at":"2026-08-06T18:47:22.629414Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:22.792944Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:22.792944Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:e13d3914d1e0bff3b2e88b9392132b7c821b3f9dc58255f4324fce6e0820b844","observation_id":"6659a5b5-b057-4067-bf44-5df7e27399e9","resolution":{"observed_at":"2026-08-06T18:47:22.792944Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:24.059954Z","title":null,"venue":null,"work_id":"ba6e3a2f-486b-44b3-9e4a-6be2b68ddf0d","year":2023},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:22.943545Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:69ee0ed21d7e35f12ddb6780e4b444604975b1553260d08c6c2fb8757855a34c","observation_id":"bfaa7b3a-f669-4c3d-badf-bb1c07864b51","resolution":{"observed_at":"2026-08-06T18:47:24.063051Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.11425","last_updated":"2025-03-24T10:18:56Z","snapshot_observed_at":"2026-07-06T20:23:22.709846Z","submitted_at":"2025-01-20T11:46:04Z","title":"Agent-R: Training Language Model Agents to Reflect via Iterative Self-Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.11425","snapshot_observed_at":"2026-08-06T18:47:23.023790Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:23.023790Z"},"links":{"cited_paper":"/paper/2501.11425","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:67456a66a8f1c29b8b28d207224cb93cafee0436f689102b6e96079e6d748634","observation_id":"5e784283-0055-4fc5-9ae7-06df5910fdd3","resolution":{"observed_at":"2026-08-06T18:47:23.023790Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.01825","last_updated":"2023-09-13T03:57:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-03T15:34:01Z","title":"Scaling Relationship on Learning Mathematical Reasoning with Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.01825","snapshot_observed_at":"2026-08-06T18:47:23.075419Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:23.075419Z"},"links":{"cited_paper":"/paper/2308.01825","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:eb52aa7935d1772e60c2273355251bd9acb6bfed627ec66dc5ece11723688537","observation_id":"1bd3cfd6-4049-470e-af4d-e5c63b7094da","resolution":{"observed_at":"2026-08-06T18:47:23.075419Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:24.050262Z","title":null,"venue":null,"work_id":"26b2138e-1e57-473d-b6bb-5990278409bf","year":2022},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:23.191916Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:728ecab420542e8c249df9fb4f92b94ca50852b44c1ab2d02851503a9a74997b","observation_id":"e4efc397-8e73-4806-ab5e-4ab405074120","resolution":{"observed_at":"2026-08-06T18:47:24.053599Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:23.310760Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:23.310760Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:42da2496dbafc2ccd4b597c0b788b3c4ce63b8010b27351295440020cd839598","observation_id":"55f2cdf5-e890-4e21-bc6b-14ac32574dba","resolution":{"observed_at":"2026-08-06T18:47:23.310760Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:24.012163Z","title":null,"venue":null,"work_id":"e237108a-2e93-42ee-b140-54d64e448955","year":2025},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:23.322550Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:2cb466806fa5ddb2a5c5cc891a09b0d37e84c581f3413d9c9ce216aac8b3becf","observation_id":"a3b23f49-3da8-4173-9fa1-fe4e5894c6aa","resolution":{"observed_at":"2026-08-06T18:47:24.041514Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09959","last_updated":"2025-01-17T05:21:49Z","snapshot_observed_at":"2026-07-06T20:22:19.281265Z","submitted_at":"2025-01-17T05:21:49Z","title":"A Survey on Multi-Turn Interaction Capabilities of Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.09959","snapshot_observed_at":"2026-08-06T18:47:23.325662Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:23.325662Z"},"links":{"cited_paper":"/paper/2501.09959","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:4f2630bd758e194c3152cbfbcba711277d5c36c79c9c6799e535b77fe3a0b6bd","observation_id":"bff288c3-8dd9-4cdd-a884-46bef3babba4","resolution":{"observed_at":"2026-08-06T18:47:23.325662Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:23.789171Z","title":null,"venue":null,"work_id":"339e5907-04e6-4447-b743-437403372636","year":2025},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:23.367773Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:4778bfc08437917153a0c1219118ec05700d84f31041097d7975482f602c7cf7","observation_id":"7da4d6c5-15f8-4e26-8e69-46c479e6fc78","resolution":{"observed_at":"2026-08-06T18:47:23.992294Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-09T10:40:54.885700Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation"},"reference_resolution":{"displayed":39,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":38,"verified_exact":1,"verified_fuzzy":0},"total_outbound_references":39},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 39 of 39 outbound references and 0 inbound Pith citation observations for arXiv:2507.07441."}