{"as_of":"2026-08-10T17:10:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:331702abdbbef5d1ed0111ce022d580faaf397a1e169f6458e31de26dea5bd6f","coverage":[{"denominator":41,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":41,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-03T00:13:30.397515Z","state":"measured"},{"denominator":41,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":41,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2602.11351/citation-record","integrity":"/paper/2602.11351/integrity","json":"/paper/2602.11351/citation-record.json","paper":"/paper/2602.11351"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T00:13:24.996133Z","title":"Consistently simulating human personas with multi-turn reinforcement learning.arXiv preprint arXiv:2511.00222,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:24.996133Z"},"links":{"citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:bc9ec28205a785e92e02a0423c943347ad456d6b88ae9d33f1f105b4766146e6","observation_id":"4cf34756-4688-4a94-a3f3-eb8df2a296f7","resolution":{"observed_at":"2026-08-03T00:13:24.996133Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-10T12:59:15.517391Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-03T00:13:25.232880Z","title":"Reinforce- ment learning for long-horizon interactive llm agents","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:25.232880Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:57ffed03e62d34f1e9e7e81dae1b13a4472f4a3503c67c66a6eea99195293383","observation_id":"925a5ce1-d2d7-4178-bfe5-2fb732e95e7c","resolution":{"observed_at":"2026-08-03T00:13:25.232880Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01307","last_updated":"2025-08-15T15:21:46Z","snapshot_observed_at":"2026-08-08T21:52:29.510852Z","submitted_at":"2025-03-03T08:46:22Z","title":"Cognitive Behaviors that Enable Self-Improving Reasoners, or, Four Habits of Highly Effective STaRs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01307","snapshot_observed_at":"2026-08-03T00:13:25.386643Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:25.386643Z"},"links":{"cited_paper":"/paper/2503.01307","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:a823d3a0e13acfcb4094cd109f4b5af9d1342a333b0ce485ff483dcf83bf5da9","observation_id":"5d47239c-cc22-4a5b-98c7-0fd887ce84b7","resolution":{"observed_at":"2026-08-03T00:13:25.386643Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-03T00:13:25.542401Z","title":"Deepseek-r1: In- centivizing reasoning capability in llms via reinforcement learning.arXiv preprint arXiv:2501.12948,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:25.542401Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:861aa429b4f301db5361e7ecc49cbaef00b878cb368a54bafc6209ec6dea62aa","observation_id":"2f6680bd-5aac-44f1-a237-74399e3fa718","resolution":{"observed_at":"2026-08-03T00:13:25.542401Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1502.02259","last_updated":"2015-02-08T14:58:50Z","snapshot_observed_at":"2026-08-03T17:17:59.241164Z","submitted_at":"2015-02-08T14:58:50Z","title":"Contextual Markov Decision Processes","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1502.02259","snapshot_observed_at":"2026-08-03T00:13:25.745024Z","title":"Con- textual markov decision processes.arXiv preprint arXiv:1502.02259,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:25.745024Z"},"links":{"cited_paper":"/paper/1502.02259","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:76896b96506087272cd0a16efe49c4fa5a89ce371efe8a283fb1f023fef87b6f","observation_id":"de66711d-3d1d-4196-a7cb-7a50aa09c23d","resolution":{"observed_at":"2026-08-03T00:13:25.745024Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T00:13:26.242813Z","title":"R., He, J., Yu, H., et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:26.242813Z"},"links":{"citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:8f33c10c51a7a7e2e7ba242504f6e082e819e1eb323cb6220ade1460651044c3","observation_id":"d3cd077b-4b50-4363-b7c5-5ba09127493d","resolution":{"observed_at":"2026-08-03T00:13:26.242813Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T00:13:26.329894Z","title":"Quagmires in sft-rl post- training: When high sft scores mislead and what to use instead.arXiv preprint arXiv:2510.01624,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:26.329894Z"},"links":{"citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:6efe3a892bd42ec75cbd73e56609c81bf35fb6a9a3768414ac59259694564603","observation_id":"ed6abef3-9eaa-4136-af20-fc317702da7d","resolution":{"observed_at":"2026-08-03T00:13:26.329894Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.01679","last_updated":"2025-06-03T20:51:06Z","snapshot_observed_at":"2026-08-05T20:06:13.107818Z","submitted_at":"2024-10-02T15:49:30Z","title":"VinePPO: Refining Credit Assignment in RL Training of LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.01679","snapshot_observed_at":"2026-08-03T00:13:26.456107Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:26.456107Z"},"links":{"cited_paper":"/paper/2410.01679","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:0f60a0443a15adb8ff8d0c8c9da698c3dcc9ebdf597f5b5038a7d05bcadd0745","observation_id":"fdafd0cb-8359-4589-80fa-a54d0b1c4e9f","resolution":{"observed_at":"2026-08-03T00:13:26.456107Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.17428","last_updated":"2025-02-25T00:35:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-27T17:59:45Z","title":"NV-Embed: Improved Techniques for Training LLMs as Generalist Embedding Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.17428","snapshot_observed_at":"2026-08-03T00:13:26.575467Z","title":"Nv-embed: Improved techniques for training llms as generalist embedding models.arXiv preprint arXiv:2405.17428,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:26.575467Z"},"links":{"cited_paper":"/paper/2405.17428","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:a95f7494e52c1f2c952c3f69bebb838b6dfffe99dc9a5cdcb25e1727b8e70b29","observation_id":"1dbddb93-2803-4ca7-88e5-63fe4e100703","resolution":{"observed_at":"2026-08-03T00:13:26.575467Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14106","last_updated":"2025-05-25T18:28:20Z","snapshot_observed_at":"2026-08-07T23:12:15.391703Z","submitted_at":"2025-05-20T09:13:22Z","title":"A Personalized Conversational Benchmark: Towards Simulating Personalized Conversations","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.14106","snapshot_observed_at":"2026-08-03T00:13:26.701496Z","title":"A., Dernoncourt, F., Kveton, B., Wu, J., Yu, T., Song, L., Yang, T., Qin, Y ., et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:26.701496Z"},"links":{"cited_paper":"/paper/2505.14106","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:674da4b11d9314d08f4f59fa6f7910dc4da21c4c962db71adb170f8903739b9a","observation_id":"bd3124c4-1797-4d42-9b63-7bc0125cb811","resolution":{"observed_at":"2026-08-03T00:13:26.701496Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.14295","last_updated":"2025-08-22T16:49:10Z","snapshot_observed_at":"2026-08-10T14:21:15.674673Z","submitted_at":"2025-07-18T18:07:38Z","title":"A Simple \"Try Again\" Can Elicit Multi-Turn LLM Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.14295","snapshot_observed_at":"2026-08-03T00:13:26.854490Z","title":"Let’s try again: Eliciting multi-turn reasoning 9 Behavioral Agentic Optimization in language models via simplistic feedback","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:26.854490Z"},"links":{"cited_paper":"/paper/2507.14295","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:eb1d01297cf8387688f909d9da3f12ea204308b9c64394c6d726c10460376467","observation_id":"f75ca2ed-f3e2-4825-8c23-754567efa4a3","resolution":{"observed_at":"2026-08-03T00:13:26.854490Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.21620","last_updated":"2025-05-24T08:46:08Z","snapshot_observed_at":"2026-08-01T20:06:59.931737Z","submitted_at":"2025-03-27T15:39:30Z","title":"UI-R1: Enhancing Efficient Action Prediction of GUI Agents by Reinforcement Learning","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.21620","snapshot_observed_at":"2026-08-03T00:13:27.015676Z","title":"Ui-r1: Enhancing efficient action prediction of gui agents by reinforcement learning.arXiv preprint arXiv:2503.21620,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:27.015676Z"},"links":{"cited_paper":"/paper/2503.21620","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:075f219df1380776d7672092429cc7f55ac8135e46d616eafe69d5826adb8f21","observation_id":"21967e0c-e5f6-42a9-a2a2-d93f2d683900","resolution":{"observed_at":"2026-08-03T00:13:27.015676Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.10458","last_updated":"2025-10-01T04:55:20Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-14T17:45:54Z","title":"GUI-R1 : A Generalist R1-Style Vision-Language Action Model For GUI Agents","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.10458","snapshot_observed_at":"2026-08-03T00:13:27.189438Z","title":"Gui- r1: A generalist r1-style vision-language action model for gui agents.arXiv preprint arXiv:2504.10458,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:27.189438Z"},"links":{"cited_paper":"/paper/2504.10458","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:2c46789929fed44da9abf95dfd226c9e08a1cb2785331a8e1b51f3c4e0b376aa","observation_id":"c097a627-c7e7-408b-87bf-6b323c113e63","resolution":{"observed_at":"2026-08-03T00:13:27.189438Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.07152","last_updated":"2025-09-07T02:40:21Z","snapshot_observed_at":"2026-08-05T05:33:12.830903Z","submitted_at":"2024-11-11T17:28:19Z","title":"HierTOD: A Task-Oriented Dialogue System Driven by Hierarchical Goals","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.07152","snapshot_observed_at":"2026-08-03T00:13:27.347739Z","title":"Hiertod: A task-oriented dialogue system driven by hier- archical goals.arXiv preprint arXiv:2411.07152,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:27.347739Z"},"links":{"cited_paper":"/paper/2411.07152","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:a8ee460bb12cad4e0d1fefc09b388f9f83ae53870abb9f7de48da185961216f3","observation_id":"ab94e218-f79c-409f-8062-ecaef6c54f67","resolution":{"observed_at":"2026-08-03T00:13:27.347739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.11827","last_updated":"2025-05-25T04:09:29Z","snapshot_observed_at":"2026-08-07T15:44:24.589199Z","submitted_at":"2025-05-17T04:26:39Z","title":"Not All Thoughts are Generated Equal: Efficient LLM Reasoning via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.11827","snapshot_observed_at":"2026-08-03T00:13:27.574670Z","title":"Not all thoughts are generated equal: Efficient llm reasoning via multi-turn reinforcement learning.arXiv preprint arXiv:2505.11827,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:27.574670Z"},"links":{"cited_paper":"/paper/2505.11827","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:c49b3e020c18312838afb5a5f4fb65ab165b2dd542a922cf274afae29e18ff29","observation_id":"a8a8e0be-0c63-4936-a4fb-ca7f1444b804","resolution":{"observed_at":"2026-08-03T00:13:27.574670Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-08-07T21:22:03.988482Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-08-03T00:13:27.793264Z","title":"Balrog: Benchmarking agentic llm and vlm reasoning on games.arXiv preprint arXiv:2411.13543,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:27.793264Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:d2978b13bce95205e56d08782810d6c3d0e221b640e822de79d0f58f546c296f","observation_id":"97c2fc39-6466-4cf3-a527-32042fbe3344","resolution":{"observed_at":"2026-08-03T00:13:27.793264Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.03601","last_updated":"2025-07-19T17:39:17Z","snapshot_observed_at":"2026-08-07T20:27:54.118684Z","submitted_at":"2025-04-04T17:13:57Z","title":"APIGen-MT: Agentic Pipeline for Multi-Turn Data Generation via Simulated Agent-Human Interplay","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.03601","snapshot_observed_at":"2026-08-03T00:13:27.863872Z","title":"C., et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:27.863872Z"},"links":{"cited_paper":"/paper/2504.03601","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:f8350f489cd4ba9fdaebcb22eed357b20e75bcfaba29aa5c918becb0f829f2ea","observation_id":"670737dc-3c3a-42ab-a041-0bf224cc8170","resolution":{"observed_at":"2026-08-03T00:13:27.863872Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.13958","last_updated":"2025-04-16T21:45:32Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-16T21:45:32Z","title":"ToolRL: Reward is All Tool Learning Needs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.13958","snapshot_observed_at":"2026-08-03T00:13:27.985409Z","title":"C., He, Q., Wang, H., Chen, X., Hakkani-T¨ur, D., Tur, G., and Ji, H","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:27.985409Z"},"links":{"cited_paper":"/paper/2504.13958","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:21e63423c3c03f8ea199d3679bbd44778d63f0f406e6cba3ea4b0cec86cad651","observation_id":"1f86f967-51b6-442c-a056-e36475a4f012","resolution":{"observed_at":"2026-08-03T00:13:27.985409Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-03T00:13:28.112926Z","title":"Deepseekmath: Push- ing the limits of mathematical reasoning in open language models.arXiv preprint arXiv:2402.03300,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:28.112926Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:50d04266f9391a889b8e9a3158a04e82cfcfeea0c6bf25737a8323f1da24ee10","observation_id":"df72b7ae-2db8-4bda-985b-47de9f1a2166","resolution":{"observed_at":"2026-08-03T00:13:28.112926Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.02508","last_updated":"2025-06-16T03:29:47Z","snapshot_observed_at":"2026-08-09T11:50:28.856427Z","submitted_at":"2025-02-04T17:26:58Z","title":"Satori: Reinforcement Learning with Chain-of-Action-Thought Enhances LLM Reasoning via Autoregressive Search","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.02508","snapshot_observed_at":"2026-08-03T00:13:28.211988Z","title":"Satori: Reinforcement learning with chain-of-action-thought en- hances llm reasoning via autoregressive search.arXiv preprint arXiv:2502.02508,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:28.211988Z"},"links":{"cited_paper":"/paper/2502.02508","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:16e9d1f7147a57850d84c0aa22ee98bce470572e6512719d16cdc094a053f58e","observation_id":"2465e9b8-0ca7-4897-9838-026eacf3cd3d","resolution":{"observed_at":"2026-08-03T00:13:28.211988Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06358","last_updated":"2025-03-08T23:41:20Z","snapshot_observed_at":"2026-08-07T17:19:44.883953Z","submitted_at":"2025-03-08T23:41:20Z","title":"Language Model Personalization via Reward Factorization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06358","snapshot_observed_at":"2026-08-03T00:13:28.352379Z","title":"Language model personalization via reward factorization","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:28.352379Z"},"links":{"cited_paper":"/paper/2503.06358","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:7ff8608caa773d3e0d5aba50ceb2dae6813d29593bc64933786c6257d9758dbf","observation_id":"fbaf1417-b5da-400e-9a9f-40f7dd5c0742","resolution":{"observed_at":"2026-08-03T00:13:28.352379Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.01441","last_updated":"2025-04-28T10:42:49Z","snapshot_observed_at":"2026-08-05T20:36:54.358831Z","submitted_at":"2025-04-28T10:42:49Z","title":"Agentic Reasoning and Tool Integration for LLMs via Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.01441","snapshot_observed_at":"2026-08-03T00:13:28.480694Z","title":"Agentic reasoning and tool integration for llms via reinforcement learning.arXiv preprint arXiv:2505.01441,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:28.480694Z"},"links":{"cited_paper":"/paper/2505.01441","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:e015d0882a63ffa320c700da83d2f7a41e7f96bc2c67d8a16016e500925c24a3","observation_id":"fc728a6f-6756-4464-bffe-ece9910977f7","resolution":{"observed_at":"2026-08-03T00:13:28.480694Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T00:13:28.610154Z","title":"Training proactive and per- sonalized llm agents.arXiv preprint arXiv:2511.02208,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:28.610154Z"},"links":{"citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:b410d52b58b7935bcd25702d0c342eb61e60231dcf3d791583605c633dc32f6f","observation_id":"d746fc4d-08f7-4f89-804f-0866a12d889f","resolution":{"observed_at":"2026-08-03T00:13:28.610154Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-03T00:13:28.684575Z","title":"M., Hauth, A., Millican, K., et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:28.684575Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:5bcf66195021503c858963ca3e68ab9c74050b1b95feca316b150ae29caa5189","observation_id":"0b738638-e145-4a6d-a552-07f26960a010","resolution":{"observed_at":"2026-08-03T00:13:28.684575Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T00:13:28.984404Z","title":"En- hancing personalized multi-turn dialogue with curiosity reward.arXiv preprint arXiv:2504.03206,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:28.984404Z"},"links":{"citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:7331840c2febbce4b463c864edee27fd82a899992d5ffb0628b97947f6b76ea2","observation_id":"10028b63-61cd-4504-8a77-e75ff9eb10a6","resolution":{"observed_at":"2026-08-03T00:13:28.984404Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05606","last_updated":"2026-05-17T17:20:46Z","snapshot_observed_at":"2026-08-06T09:19:52.858868Z","submitted_at":"2025-06-05T21:37:49Z","title":"OPeRA: A Dataset of Observation, Persona, Rationale, and Action for Evaluating LLMs on Human Online Shopping Behavior Simulation","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.05606","snapshot_observed_at":"2026-08-03T00:13:29.080919Z","title":"Opera: A dataset of observation, persona, rationale, and action for evaluating llms on human online shopping behavior simulation.arXiv preprint arXiv:2506.05606, 2025a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:29.080919Z"},"links":{"cited_paper":"/paper/2506.05606","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:a829e51879fe6415a3616f3c651876a82e6b2fa80c8b50d9678e3b2efaad4fcf","observation_id":"9391ca7b-429b-4994-a220-a8f2c6f0f8b9","resolution":{"observed_at":"2026-08-03T00:13:29.080919Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T00:13:29.386885Z","title":"Boad: Discovering hierarchi- cal software engineering agents via bandit optimization","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:29.386885Z"},"links":{"citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:b9f4ed361e2aa21b421d75066e3711ad1331c1e35f274b947bd954fdb3313791","observation_id":"b279e7df-c54c-4fe5-a995-6298419c2e5b","resolution":{"observed_at":"2026-08-03T00:13:29.386885Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-08-03T00:13:29.423336Z","title":"Qwen3 technical report.arXiv preprint arXiv:2505.09388, 2025a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:29.423336Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:aa13bda4847c53765952ace7a9755f70cb533b12207ce476a427f69d6d83e1aa","observation_id":"43f0cd01-802e-484c-b23a-15c30fe2faac","resolution":{"observed_at":"2026-08-03T00:13:29.423336Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.03373","last_updated":"2025-02-05T17:13:32Z","snapshot_observed_at":"2026-07-06T20:31:41.231839Z","submitted_at":"2025-02-05T17:13:32Z","title":"Demystifying Long Chain-of-Thought Reasoning in LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.03373","snapshot_observed_at":"2026-08-03T00:13:29.528540Z","title":"Demys- tifying long chain-of-thought reasoning in llms.arXiv preprint arXiv:2502.03373,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:29.528540Z"},"links":{"cited_paper":"/paper/2502.03373","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:c1ef8dffb9a79c7a4c34f5be0f8af70b084cbc27f177ca87edb891d2bf3bc551","observation_id":"3bd0f250-4516-44ba-9c1a-1f3650c12246","resolution":{"observed_at":"2026-08-03T00:13:29.528540Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T00:13:29.699949Z","title":"Demysti- fying reinforcement learning in agentic reasoning.arXiv preprint arXiv:2510.11701,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:29.699949Z"},"links":{"citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:3ce9902e4ea5d99a92a4caaa7dee30111c045a47c96376d01ba53af59251e046","observation_id":"23557dd4-11bf-41ca-905a-c1c1bfb8f343","resolution":{"observed_at":"2026-08-03T00:13:29.699949Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23604","last_updated":"2025-05-29T16:15:36Z","snapshot_observed_at":"2026-08-09T05:03:49.842579Z","submitted_at":"2025-05-29T16:15:36Z","title":"Satori-SWE: Evolutionary Test-Time Scaling for Sample-Efficient Software Engineering","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.23604","snapshot_observed_at":"2026-08-03T00:13:29.838143Z","title":"Satori- swe: Evolutionary test-time scaling for sample-efficient software engineering.arXiv preprint arXiv:2505.23604, 2025a","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:29.838143Z"},"links":{"cited_paper":"/paper/2505.23604","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:f4c499c17bff188ff8e8000e7367a7240bcdddc4c14a49e556e16295cca24e34","observation_id":"91f78ddb-4cbf-4659-9b2a-0d815ea1ca68","resolution":{"observed_at":"2026-08-03T00:13:29.838143Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T00:13:30.021924Z","title":"Teaching language models to evolve with users: Dynamic profile modeling for personalized alignment.arXiv preprint arXiv:2505.15456,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:30.021924Z"},"links":{"citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:0368d6131009ae1047872cfb604d1c5d22a796c79d029a049fa2d255d9bb3bb1","observation_id":"10def8fe-8f92-4e75-af87-e71342188605","resolution":{"observed_at":"2026-08-03T00:13:30.021924Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.13372","last_updated":"2024-06-27T22:44:48Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-20T08:08:54Z","title":"LlamaFactory: Unified Efficient Fine-Tuning of 100+ Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.13372","snapshot_observed_at":"2026-08-03T00:13:30.206953Z","title":"L., Huang, J., Yu, C","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:30.206953Z"},"links":{"cited_paper":"/paper/2403.13372","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:833df93c99c5de81a0a8fc37f103218f8533bd92cf69c5e78d0228a0e5c4b965","observation_id":"e3738c32-c27f-4543-aa82-ea31e4711187","resolution":{"observed_at":"2026-08-03T00:13:30.206953Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T00:13:30.321333Z","title":"E., and Zhou, W","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:30.321333Z"},"links":{"citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:a39676aeaa7d0411969319e3d0aee8fd867a9962c8d45c9e5397b77a7a5dda59","observation_id":"9a4cb23e-93d5-4e7a-b7a3-f78585c61635","resolution":{"observed_at":"2026-08-03T00:13:30.321333Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T00:13:30.397515Z","title":"Yes”, “No","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:30.397515Z"},"links":{"citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:e818b7931593f490b2178daec5701976c21d0d04e017b22d61270c65a4f85287","observation_id":"0f994dda-d6a8-42fd-964a-a8595da18906","resolution":{"observed_at":"2026-08-03T00:13:30.397515Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-03T00:13:25.873768Z","title":"P., Perelman, A., Ramesh, A., Clark, A., Ostrow, A., Welihinda, A., Hayes, A., Radford, A., et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":2015,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:25.873768Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:a691611fd3fd7cb218b1696e10cda8f2c423a20639528cd80e0e11f1342069d8","observation_id":"5a0a8997-cc09-4a0b-a013-6621dda5a016","resolution":{"observed_at":"2026-08-03T00:13:25.873768Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.25140","last_updated":"2026-03-16T20:49:28Z","snapshot_observed_at":"2026-08-02T12:08:17.149184Z","submitted_at":"2025-09-29T17:51:03Z","title":"ReasoningBank: Scaling Agent Self-Evolving with Reasoning Memory","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.25140","snapshot_observed_at":"2026-08-03T00:13:27.661295Z","title":"T., Daruki, S., Tang, X., et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:27.661295Z"},"links":{"cited_paper":"/paper/2509.25140","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:e7882af40e73a1ca225997fdef9a6441f8dbd9035cb4aa1cf1923bb6d33d8738","observation_id":"11359bac-b2ad-4c72-8e1a-64e34cd6f14f","resolution":{"observed_at":"2026-08-03T00:13:27.661295Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.17032","last_updated":"2025-11-02T13:42:19Z","snapshot_observed_at":"2026-07-06T18:51:06.750136Z","submitted_at":"2024-07-24T06:35:05Z","title":"Gymnasium: A Standard Interface for Reinforcement Learning Environments","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.17032","snapshot_observed_at":"2026-08-03T00:13:28.882418Z","title":"U., De Cola, G., Deleu, T., Goul˜ao, M., Kallinteris, A., Krimmel, M., KG, A., et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:28.882418Z"},"links":{"cited_paper":"/paper/2407.17032","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:de2fa7816c6d07cca219e5dd59c40e7fcc6bf841bccf04d43819f1a7e06f31c0","observation_id":"8ce43951-02aa-4c32-9da1-fd317ea5175d","resolution":{"observed_at":"2026-08-03T00:13:28.882418Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16720","last_updated":"2026-04-30T02:46:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-21T18:04:31Z","title":"OpenAI o1 System Card","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.16720","snapshot_observed_at":"2026-08-03T00:13:26.072593Z","title":"Openai o1 system card.arXiv preprint arXiv:2412.16720,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:26.072593Z"},"links":{"cited_paper":"/paper/2412.16720","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:d53cb851217455e353b07c2f547897979267723303f3b3cd4f7cd450a3c07e20","observation_id":"d3b8c592-c102-4c97-a2f1-41c7335db57c","resolution":{"observed_at":"2026-08-03T00:13:26.072593Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T00:13:25.140784Z","title":"Behavior injection: Preparing language models for reinforcement learning.arXiv preprint arXiv:2505.18917,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:25.140784Z"},"links":{"citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:ebb040a90b98a8120c64bb32642f82137a20c3bdc1121f65c2a283484352e3d7","observation_id":"f09582dc-a9eb-4fe9-9b35-23ec975f43e3","resolution":{"observed_at":"2026-08-03T00:13:25.140784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.18449","last_updated":"2025-12-01T00:16:59Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-25T18:45:04Z","title":"SWE-RL: Advancing LLM Reasoning via Reinforcement Learning on Open Software Evolution","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.18449","snapshot_observed_at":"2026-08-03T00:13:29.208730Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":2026,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:29.208730Z"},"links":{"cited_paper":"/paper/2502.18449","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:ac87c245590b5be7fd88388f6afa468ad67a4b135986484a469a40e3ed6fb526","observation_id":"a279625d-8e07-44d8-a331-6aec4580110f","resolution":{"observed_at":"2026-08-03T00:13:29.208730Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","latest_version":2,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-08T10:00:22.672311Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization"},"reference_resolution":{"displayed":41,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":41,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":41},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 41 of 41 outbound references and 0 inbound Pith citation observations for arXiv:2602.11351."}