{"as_of":"2026-08-09T20:10:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c1b89eecf72a5965facb6e03e8aa4b7e9ff0e84573d53349cb2dc139fd72bdeb","coverage":[{"denominator":102,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T01:12:39.559055Z","state":"measured"},{"denominator":100,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":100,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.00155/citation-record","integrity":"/paper/2608.00155/integrity","json":"/paper/2608.00155/citation-record.json","paper":"/paper/2608.00155"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:30.678535Z","title":"A survey of self-evolving agents: What, when, how, and where to evolve on the path to artificial super intelligence.Transactions on Machine Learning Research, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:30.678535Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:84f4b813edb572641fb4e5c2c4cc7436acbf386afef6771c9496536be78728dc","observation_id":"67b611d2-1277-424a-96f0-bd6abc117c0c","resolution":{"observed_at":"2026-08-04T01:12:30.678535Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:30.740362Z","title":"Position: Agentic evolution is the path to evolving llms.arXiv preprint arXiv:2602.00359, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:30.740362Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:f8e2067650680d32bf104be9a3b764487b8062e01f011cb8fde43d6e389d9169","observation_id":"15c3b5f3-0cc4-407d-97dd-53e6f149a015","resolution":{"observed_at":"2026-08-04T01:12:30.740362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.07407","last_updated":"2025-08-31T14:55:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-10T16:07:32Z","title":"A Comprehensive Survey of Self-Evolving AI Agents: A New Paradigm Bridging Foundation Models and Lifelong Agentic Systems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.07407","snapshot_observed_at":"2026-08-04T01:12:30.844983Z","title":"A comprehensive survey of self-evolving ai agents: A new paradigm bridging foundation models and lifelong agentic systems.arXiv preprint arXiv:2508.07407, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:30.844983Z"},"links":{"cited_paper":"/paper/2508.07407","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:22b46ee7fe17643dccc84430a13ed76b0b2f3f4ee2c552f28f3786852575c1ff","observation_id":"774e17c1-81ba-4c06-8184-c3942a218a0c","resolution":{"observed_at":"2026-08-04T01:12:30.844983Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:30.945741Z","title":"Agentic context engineering: Evolving contexts for self-improving language models","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:30.945741Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:3aa2f65bd596aa34cd82f86a886cdb0ed62b9720bbfa14ad42060710cbdcc75b","observation_id":"f6d0b54b-6b3b-4420-ba41-a3bb96d97d7c","resolution":{"observed_at":"2026-08-04T01:12:30.945741Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.30621","last_updated":"2026-05-28T22:16:14Z","snapshot_observed_at":"2026-08-02T11:09:20.796647Z","submitted_at":"2026-05-28T22:16:14Z","title":"Harness Updating Is Not Harness Benefit: Disentangling Evolution Capabilities in Self-Evolving LLM Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.30621","snapshot_observed_at":"2026-08-04T01:12:31.047941Z","title":"Harness updating is not harness benefit: Disentangling evolution capabilities in self-evolving llm agents.arXiv preprint arXiv:2605.30621, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:31.047941Z"},"links":{"cited_paper":"/paper/2605.30621","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:489017180138e8e9ccb232c2963a9833c7c2aa4c384a135b386c87f0460a0577","observation_id":"4b30595c-b4bf-4321-82f1-c1ac004ba677","resolution":{"observed_at":"2026-08-04T01:12:31.047941Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:31.168633Z","title":"A-mem: Agentic memory for llm agents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:31.168633Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:6073d22bc8c7dedde578e97d2450f87ae4bdf410cc2aa950363255d3eb4df195","observation_id":"4d1cbd62-4880-457e-9d8e-76e4c51ce49d","resolution":{"observed_at":"2026-08-04T01:12:31.168633Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.03192","last_updated":"2026-02-12T05:43:57Z","snapshot_observed_at":"2026-08-08T14:48:20.406898Z","submitted_at":"2026-01-06T17:14:50Z","title":"MemRL: Self-Evolving Agents via Runtime Reinforcement Learning on Episodic Memory","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.03192","snapshot_observed_at":"2026-08-04T01:12:31.238737Z","title":"Memrl: Self-evolving agents via runtime reinforcement learning on episodic memory","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:31.238737Z"},"links":{"cited_paper":"/paper/2601.03192","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:289c5e6d6d45cb8e315dfb94cc51ed0df046d07d417b9069b638583a7865c8f6","observation_id":"5b4a68ac-830d-4563-8af2-bc2c42882100","resolution":{"observed_at":"2026-08-04T01:12:31.238737Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:31.339997Z","title":"Autoskill: Experience-driven lifelong learning via skill self-evolution.arXiv preprint arXiv:2603.01145, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:31.339997Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:06e1e260fabb38821904d1a779c8d64ab2997802a4ae86628f563ed79cc9dd4a","observation_id":"9997acb5-415b-4035-aced-f5f12b482392","resolution":{"observed_at":"2026-08-04T01:12:31.339997Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:31.448963Z","title":"Le, Samira Daruki, Xiangru Tang, Vishy Tirumalashetty, George Lee, Mahsan Rofouei, Hangfei Lin, Jiawei Han, Chen-Yu Lee, and Tomas Pfister","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:31.448963Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:b00b4e2b51a0e128c739f3d9da4ee8a0711f991b043b1489d868bf5221ecf6cf","observation_id":"2ae5e5f3-dae5-48a1-9f88-e580831d52be","resolution":{"observed_at":"2026-08-04T01:12:31.448963Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:31.553680Z","title":"Memento-skills: Let agents design agents","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:31.553680Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:8161437c55bf941b0edf92c25c79eae2865d58aeddbb1c25cdf4950787b3e526","observation_id":"669b2ad1-81a2-4075-bd74-1379c5d0b903","resolution":{"observed_at":"2026-08-04T01:12:31.553680Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.25850","last_updated":"2026-05-18T15:11:47Z","snapshot_observed_at":"2026-07-06T23:11:39.318906Z","submitted_at":"2026-04-28T16:55:02Z","title":"Agentic Harness Engineering: Observability-Driven Automatic Evolution of Coding-Agent Harnesses","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.25850","snapshot_observed_at":"2026-08-04T01:12:31.613003Z","title":"Agentic harness engineering: Observability-driven automatic evolution of coding-agent harnesses.arXiv preprint arXiv:2604.25850, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:31.613003Z"},"links":{"cited_paper":"/paper/2604.25850","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:012f714bf983c5872cab6498b1c11a1f87e76fe3dc47102c7b29feba0f13275d","observation_id":"aff5ba51-e69a-4fe0-958c-0ca5c67e030e","resolution":{"observed_at":"2026-08-04T01:12:31.613003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:31.709243Z","title":"Appworld: A controllable world of apps and people for benchmarking interactive coding agents","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:31.709243Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:380276e5e37876c4a9fba77ea854f383d8de25ffc1f0bd1e55c5e5fd141d8a1d","observation_id":"8650ca55-dd31-4e2b-b6c2-2e8233c3b8e2","resolution":{"observed_at":"2026-08-04T01:12:31.709243Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:31.826496Z","title":"Gonzalez","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:31.826496Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:b9ac1efa5f9daf3eec9a334dd31f810ad11e366f93a650afc7378ecdbf51d10b","observation_id":"0313bafd-50e0-4c88-95b5-24fa3f54c936","resolution":{"observed_at":"2026-08-04T01:12:31.826496Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:31.918060Z","title":"Swe-bench: Can language models resolve real-world github issues? In Proc","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:31.918060Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:0a60b9fc701904117bb7d9306c28c3f6e681727588c6ab0f6e8136acdbe45914","observation_id":"192ac56d-3eb2-4e4d-966f-b596268400d9","resolution":{"observed_at":"2026-08-04T01:12:31.918060Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14249","last_updated":"2026-02-20T04:23:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-24T05:27:46Z","title":"Humanity's Last Exam","version":10},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.14249","snapshot_observed_at":"2026-08-04T01:12:31.996592Z","title":"Humanity’s last exam.arXiv preprint arXiv:2501.14249, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:31.996592Z"},"links":{"cited_paper":"/paper/2501.14249","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:c082b3d6d462b5e8f1f853e3ca8ac35dad8818ca1abb281605fa3948e4856453","observation_id":"c22fc5a9-a231-4699-b27a-21e74d18ae93","resolution":{"observed_at":"2026-08-04T01:12:31.996592Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.07982","last_updated":"2025-06-09T17:52:18Z","snapshot_observed_at":"2026-07-06T21:39:13.304260Z","submitted_at":"2025-06-09T17:52:18Z","title":"$\\tau^2$-Bench: Evaluating Conversational Agents in a Dual-Control Environment","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.07982","snapshot_observed_at":"2026-08-04T01:12:32.087301Z","title":"τ 2- Bench: Evaluating Conversational Agents in a Dual-Control Environment.arXiv preprint arXiv:2506.07982, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.087301Z"},"links":{"cited_paper":"/paper/2506.07982","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:a0161d9fd11cfe4e95c255dcf13672316a270c00ee732ab088cfff0bfeb933a5","observation_id":"45ed4434-2e42-4aac-a302-79bb5630afb5","resolution":{"observed_at":"2026-08-04T01:12:32.087301Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:32.156197Z","title":"Stream- bench: Towards benchmarking continuous improvement of language agents","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.156197Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:8848bbeb0142be401bdf015d1a60bb67b073067e99c9198b441f5f4ec2dc1ac1","observation_id":"0ab7ed43-9bc6-413b-8b5e-082f746f58f4","resolution":{"observed_at":"2026-08-04T01:12:32.156197Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2511.20857","last_updated":"2026-05-18T16:18:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-11-25T21:08:07Z","title":"Evo-Memory: Benchmarking LLM Agent Test-time Learning with Self-Evolving Memory","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2511.20857","snapshot_observed_at":"2026-08-04T01:12:32.201860Z","title":"Chi, Chi Wang, Shuo Chen, Fernando Pereira, Wang-Cheng Kang, and Derek Zhiyuan Cheng","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.201860Z"},"links":{"cited_paper":"/paper/2511.20857","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:42b52d1d18a1203cc55d971b0c28d82fa756e43600a28bdb0a554f78556f37d1","observation_id":"19de0ee3-c2cd-4dac-84f9-1710b859a36c","resolution":{"observed_at":"2026-08-04T01:12:32.201860Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.03267","last_updated":"2026-05-01T23:55:43Z","snapshot_observed_at":"2026-08-02T10:52:10.211700Z","submitted_at":"2025-12-19T07:05:38Z","title":"OpenAI GPT-5 System Card","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.03267","snapshot_observed_at":"2026-08-04T01:12:32.267699Z","title":"Openai gpt-5 system card.arXiv preprint arXiv:2601.03267, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.267699Z"},"links":{"cited_paper":"/paper/2601.03267","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:f6e05ee26ecbdd4d73cdbff90e4a75dddf530605cd76d2ced30bd5fcd9e882fc","observation_id":"b8cc89d3-74d9-495a-a8d0-64d85f8c11ec","resolution":{"observed_at":"2026-08-04T01:12:32.267699Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:32.359163Z","title":"Gemini 3.1 Pro model card, February 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.359163Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:f684ae23aef1ceb29a1ee9db134b591fb441b9d003be141ccd404ac758c1b388","observation_id":"5ba34156-2f11-4e2f-be16-4be50f196bea","resolution":{"observed_at":"2026-08-04T01:12:32.359163Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:32.441241Z","title":"Introducing Claude Opus 4.7, April 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.441241Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:732ed95966bd379ec16e3dbf26aa93ec0f9f9d1a658df457975fa52615191152","observation_id":"b9ba58dd-a90c-4480-9b54-3e6fb70502a1","resolution":{"observed_at":"2026-08-04T01:12:32.441241Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:32.517774Z","title":"Browsecomp-plus: A more fair and transparent evaluation benchmark of deep-research agent","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.517774Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:79b20dd6cf22d69d60158e3fa1416bdbaaede1ba4d0248c20c1b1381f4352549","observation_id":"1e4e000b-6c7e-4e9c-b814-b4951cdeda99","resolution":{"observed_at":"2026-08-04T01:12:32.517774Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:32.557621Z","title":"Test-time training with self-supervision for generalization under distribution shifts","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.557621Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:38620bb301db21ec9ff14cdb5ff57a9a90862618c96ba9a2e4e3bb714f208cff","observation_id":"a9c9edd8-577d-471a-81e4-899bf8f7293c","resolution":{"observed_at":"2026-08-04T01:12:32.557621Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:32.616665Z","title":"Do we really need to access the source data? source hypothesis transfer for unsupervised domain adaptation","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.616665Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:823ec3c17b1c52a01ce09cd95f6f70e7167fb7dcbb311d3b03357e91f0a21abf","observation_id":"1ef7ec98-9596-452d-b766-b5a37395abac","resolution":{"observed_at":"2026-08-04T01:12:32.616665Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:32.706158Z","title":"Overcoming catastrophic forgetting in neural networks.Proceedings of the national academy of sciences, 114(13):3521–3526, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.706158Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:eb3d97b17a73eb6ab50e9b368e640a971472c556668b5041cf419886153e54a6","observation_id":"4ec2d7e4-606f-45df-bcbf-08412cdd5a1a","resolution":{"observed_at":"2026-08-04T01:12:32.706158Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:32.785239Z","title":"Gradient episodic memory for continual learning","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.785239Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:a316bc71e67c8cc8678e8acad2977741b130b9cc7cde43a76fb2cf5d61369cc6","observation_id":"7128a647-c5e2-4fc7-bee4-44afb379adaa","resolution":{"observed_at":"2026-08-04T01:12:32.785239Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:32.834609Z","title":"Scaling llm test-time compute optimally can be more effective than scaling model parameters","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.834609Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:fa3f593e41157f391fcf9fdceffbf46d964acd3e11e43304d048c9a76fccefee","observation_id":"13b29ec8-639f-4f57-9e71-ec2d6c5b41ed","resolution":{"observed_at":"2026-08-04T01:12:32.834609Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:32.907311Z","title":"Test-time training on nearest neighbors for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.907311Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:ea851c0a790317eed2c870e7e37174a97b0adf1dad03930c3fbb671108f433f8","observation_id":"ea24b3aa-c55c-429d-8f36-4f6fd65c80e7","resolution":{"observed_at":"2026-08-04T01:12:32.907311Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:32.947220Z","title":"Efficiently learning at test-time: Active fine-tuning of llms","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.947220Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:3ddc22034b488a4692d4b2028a88d6e18ee5d976de909f33b128f834d5d37fc9","observation_id":"69b76d30-5c03-4755-a817-87bedd5539bf","resolution":{"observed_at":"2026-08-04T01:12:32.947220Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:32.990733Z","title":"The surprising effectiveness of test-time training for few-shot learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.990733Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:c80b8148548cad3c5eeda6af171e3d91d64eff246af329df74754b66ae2377ab","observation_id":"8353d4e8-7813-4986-b32e-8b3fd29c1545","resolution":{"observed_at":"2026-08-04T01:12:32.990733Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:33.052856Z","title":"In-place test-time training","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:33.052856Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:672eae99d72f595fe23f984a584c0c88216e91e5bbc23a9769050db3730b0b74","observation_id":"1aaae17b-f559-45b8-b4aa-042734c11429","resolution":{"observed_at":"2026-08-04T01:12:33.052856Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:33.147161Z","title":"Test-time adaptation for llm agents via environment interaction","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:33.147161Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:5a2eb212fbe5d738e21fdd6b4094dd96adab82158e563bd81ac7d04bd9d91821","observation_id":"78bb23fd-0340-46e7-a979-5c9cb6c60f57","resolution":{"observed_at":"2026-08-04T01:12:33.147161Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:33.266388Z","title":"Test-time learning for large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:33.266388Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:b0107d6bc328cafa1e3fb805fa421e44749d13f0e6ee9e6ce2313f1e35a90c21","observation_id":"8c6d34b6-56e8-42fd-be9c-40b104d6fb5b","resolution":{"observed_at":"2026-08-04T01:12:33.266388Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:33.347224Z","title":"Ttrl: Test-time reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:33.347224Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:4e5980c2338e382ed194ddcd856dd9f604324b6627fcd90c830a48b32ef7de34","observation_id":"31c98bfe-fc30-4da1-b482-26138dfd8ce1","resolution":{"observed_at":"2026-08-04T01:12:33.347224Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:33.407176Z","title":"Learn- ing on the job: Test-time curricula for targeted reinforcement learning.arXiv preprint arXiv:2510.04786, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:33.407176Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:0d029c0b503c7cd9fa48d2368cc8a67b3bd9ddc33a24dc019645c4c5f4c0efb6","observation_id":"560c579e-c5e5-4904-893b-432431fa2cb1","resolution":{"observed_at":"2026-08-04T01:12:33.407176Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:33.486960Z","title":"Learning to discover at test time","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:33.486960Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:93c42da4161a73c39e92e5589f12185480158a6c3319b6895b75dfabdd9fa20b","observation_id":"729ca982-0b5d-48a6-bc45-3aa1aa448f4f","resolution":{"observed_at":"2026-08-04T01:12:33.486960Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:33.584703Z","title":"Collaborative multi-agent test-time reinforcement learning for reasoning.arXiv preprint arXiv:2601.09667, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:33.584703Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:f4f7f4e45a2fd1a6ffc16131a58cfdaf62af14f3167569ff5dd1b2ce1bd193b5","observation_id":"0dbd5315-bd55-49ce-a095-a6213b44dc44","resolution":{"observed_at":"2026-08-04T01:12:33.584703Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:33.672842Z","title":"What if consensus lies? selective-complementary reinforcement learning at test time","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:33.672842Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:19a88746e96738c742e75f8405c016987faf2c0a28fba72aff4c82449bce3b04","observation_id":"8a992a21-8df4-4314-91b8-8d6506facb9c","resolution":{"observed_at":"2026-08-04T01:12:33.672842Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:33.783687Z","title":"Ttsr: Test-time self-reflection for continual reasoning improvement.arXiv preprint arXiv:2603.03297, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:33.783687Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:1dd0d9bbb029d29e2e949adbcb446574f551820908a170ffda92d1b484c0e19e","observation_id":"0cb42fec-7ea7-47d8-add2-4a5c603aa6c1","resolution":{"observed_at":"2026-08-04T01:12:33.783687Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:33.899505Z","title":"Ttcs: Test-time curriculum synthesis for self-evolving.arXiv preprint arXiv:2601.22628, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:33.899505Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:d5a2f9f2fd1adbfbf18c2819294e1f3a5c99bbf30d846f8c22aa30ece967c4b9","observation_id":"d76d5b23-7867-48c6-ba76-c7b3adbf5fd3","resolution":{"observed_at":"2026-08-04T01:12:33.899505Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.14477","last_updated":"2026-07-14T21:30:34Z","snapshot_observed_at":"2026-08-02T14:06:39.249902Z","submitted_at":"2026-05-14T07:18:12Z","title":"Test-Time Learning with an Evolving Library","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.14477","snapshot_observed_at":"2026-08-04T01:12:34.009288Z","title":"Test-time learning with an evolving library.arXiv preprint arXiv:2605.14477, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:34.009288Z"},"links":{"cited_paper":"/paper/2605.14477","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:6e069ac6d3aa27fa8e9b86cd94093d5aecee8a41d1dc0da6237924733099fa61","observation_id":"b01a108e-71ff-4206-a9bc-190f1a4e571e","resolution":{"observed_at":"2026-08-04T01:12:34.009288Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.16986","last_updated":"2026-07-29T10:19:54Z","snapshot_observed_at":"2026-08-02T13:52:40.874253Z","submitted_at":"2026-05-16T13:14:15Z","title":"Skills on the Fly: Test-Time Adaptive Skill Synthesis for LLM Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.16986","snapshot_observed_at":"2026-08-04T01:12:34.114246Z","title":"Skills on the fly: Test-time adaptive skill synthesis for llm agents.arXiv preprint arXiv:2605.16986, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:34.114246Z"},"links":{"cited_paper":"/paper/2605.16986","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:334c8fcdd0ac12ef03df73bf459021ebec739e04ae61fa7480078f61077d723b","observation_id":"eb9d85b7-402a-49fc-aec2-7efb0b250981","resolution":{"observed_at":"2026-08-04T01:12:34.114246Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:34.226463Z","title":"Tarse: Test-time adaptation via retrieval of skills and experience for reasoning agents.arXiv preprint arXiv:2603.01241, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:34.226463Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:f98a4fea96d20549469792ac767d1c3c8954875fe3cd54d9705252e6abb5928f","observation_id":"4b2262f2-6162-4fac-abde-a3da7f3f3208","resolution":{"observed_at":"2026-08-04T01:12:34.226463Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:34.337320Z","title":"Agentic plan caching: Test-time memory for fast and cost-efficient llm agents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:34.337320Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:e5bbc5ffe1270d96d79bf8384ade8f0102d25b170e47947de17c7551b9cc02c2","observation_id":"26fb1c9a-50dd-4827-84ac-1aa5afaa48dd","resolution":{"observed_at":"2026-08-04T01:12:34.337320Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.03224","last_updated":"2026-06-06T09:48:09Z","snapshot_observed_at":"2026-08-06T23:53:57.800987Z","submitted_at":"2026-02-03T07:52:26Z","title":"TAME: A Trustworthy Test-Time Evolution of Agent Memory with Systematic Benchmarking","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.03224","snapshot_observed_at":"2026-08-04T01:12:34.442077Z","title":"Tame: A trustworthy test-time evolution of agent memory with systematic benchmarking.arXiv preprint arXiv:2602.03224, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:34.442077Z"},"links":{"cited_paper":"/paper/2602.03224","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:5b50bbf0916463829921c36f473595dafe7764ca43efbd876654f5d1121a9724","observation_id":"ed2c9bfc-f745-4cf6-b322-a65549161db3","resolution":{"observed_at":"2026-08-04T01:12:34.442077Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:34.540281Z","title":"Self-improving llm agents at test-time.arXiv preprint arXiv:2510.07841, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:34.540281Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:9dbe1213ce2dde5d9338d077ddd0e225c7deb2cfeb10d41dc55c35333ac35492","observation_id":"18107843-204b-4eb5-b983-e6c2028238e2","resolution":{"observed_at":"2026-08-04T01:12:34.540281Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:34.652568Z","title":"Just- in-time reinforcement learning: Continual learning in llm agents without gradient updates","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:34.652568Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:92a885120fc7b377e1534fe0ba4bc66379bad0a95a2cfdd0e1fbe49d08256b90","observation_id":"b09b02a8-d8b8-453a-b705-88ffabf8a7c3","resolution":{"observed_at":"2026-08-04T01:12:34.652568Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:34.761418Z","title":"Panini: Continual learning in token space via structured memory","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:34.761418Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:d245b01bf3b85b802453fc1c0040601905838e1b59f856e16085d2437fa85cfa","observation_id":"9f5ea4cc-4e98-4228-9cbd-eb3796899c71","resolution":{"observed_at":"2026-08-04T01:12:34.761418Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.03641","last_updated":"2026-04-11T14:07:11Z","snapshot_observed_at":"2026-07-06T22:40:56.386532Z","submitted_at":"2026-01-07T06:43:50Z","title":"Agent-Dice: Disentangling Knowledge Updates via Geometric Consensus for Agent Continual Learning","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.03641","snapshot_observed_at":"2026-08-04T01:12:34.866944Z","title":"Agent-dice: Disentangling knowledge updates via geometric consensus for agent continual learning.arXiv preprint arXiv:2601.03641, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:34.866944Z"},"links":{"cited_paper":"/paper/2601.03641","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:983914ddd3b3f0f06eb0b7647a0e61d5fa15d1b5b87ccec73822e95054692af7","observation_id":"5a6a2359-872e-42b0-bcc8-368eb0a6533b","resolution":{"observed_at":"2026-08-04T01:12:34.866944Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:34.939316Z","title":"Mssr: Memory-aware adaptive replay for continual llm fine-tuning.arXiv preprint arXiv:2603.09892, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:34.939316Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:9841a122366631aa710506746f8e56d3f7e123810fce2c81561172a4adb1bfb3","observation_id":"df12cecf-038a-4dc1-8a2b-c5be36a6d665","resolution":{"observed_at":"2026-08-04T01:12:34.939316Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:35.017911Z","title":"Learning to continually learn via meta-learning agentic memory designs.arXiv preprint arXiv:2602.07755, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:35.017911Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:3de6b14607992f2ca50b94d6df3e0177305342dd4aa94de95f6a75a95af41eb4","observation_id":"fec8e04f-7212-49fe-a4ed-8b154e68aa71","resolution":{"observed_at":"2026-08-04T01:12:35.017911Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:35.087726Z","title":"Xskill: Continual learning from experience and skills in multimodal agents","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:35.087726Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:01d197ae92bf0a56294a4638490cd12f8e7a32149e833eb7eeacef6cfb4cfa07","observation_id":"fb473b65-f10e-480b-bb9b-04a615958777","resolution":{"observed_at":"2026-08-04T01:12:35.087726Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2603.16856","last_updated":"2026-06-29T12:13:29Z","snapshot_observed_at":"2026-07-13T23:28:27.562632Z","submitted_at":"2026-03-17T17:57:49Z","title":"Online Experiential Learning for Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2603.16856","snapshot_observed_at":"2026-08-04T01:12:35.159727Z","title":"Online experiential learning for language models.arXiv preprint arXiv:2603.16856, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:35.159727Z"},"links":{"cited_paper":"/paper/2603.16856","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:4b43388598708a126292e704af853bc49278908c9ab6a66bf657df33ed7cfb91","observation_id":"da43f976-c6ce-463c-9799-361ce994f6aa","resolution":{"observed_at":"2026-08-04T01:12:35.159727Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:35.227678Z","title":"Adaptive collaboration with humans: Metacognitive policy optimization for multi-agent llms with continual learning","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:35.227678Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:86c769f26547fda6997a69a06e14e89923b77bb3d5095d0b32eeff2e90793609","observation_id":"ddc78d07-c649-491c-9b06-7196fe1fb0e7","resolution":{"observed_at":"2026-08-04T01:12:35.227678Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.02474","last_updated":"2026-05-24T19:01:09Z","snapshot_observed_at":"2026-08-03T05:25:38.558727Z","submitted_at":"2026-02-02T18:53:28Z","title":"MemSkill: Learning and Evolving Memory Skills for Self-Evolving Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.02474","snapshot_observed_at":"2026-08-04T01:12:35.336093Z","title":"Memskill: Learning and evolving memory skills for self-evolving agents.arXiv preprint arXiv:2602.02474, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:35.336093Z"},"links":{"cited_paper":"/paper/2602.02474","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:644e7de876d5b42ad0f065c688789cdbb58a03a2db87601c475b25cdf63d7bff","observation_id":"d42845b4-b90e-46aa-b1b5-2dbf86837931","resolution":{"observed_at":"2026-08-04T01:12:35.336093Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.19413","last_updated":"2025-04-28T01:46:35Z","snapshot_observed_at":"2026-08-02T07:32:11.339534Z","submitted_at":"2025-04-28T01:46:35Z","title":"Mem0: Building Production-Ready AI Agents with Scalable Long-Term Memory","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.19413","snapshot_observed_at":"2026-08-04T01:12:35.459460Z","title":"Mem0: Building production-ready ai agents with scalable long-term memory.arXiv preprint arXiv:2504.19413, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:35.459460Z"},"links":{"cited_paper":"/paper/2504.19413","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:5df54bb739e537e5a81024e3fd24135b6233d2aa28b99cdd5e0095f8650e3864","observation_id":"227b17f5-290a-48c0-94bd-9da4a8674c98","resolution":{"observed_at":"2026-08-04T01:12:35.459460Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.08234","last_updated":"2026-02-09T03:17:17Z","snapshot_observed_at":"2026-07-06T22:45:05.823859Z","submitted_at":"2026-02-09T03:17:17Z","title":"SkillRL: Evolving Agents via Recursive Skill-Augmented Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.08234","snapshot_observed_at":"2026-08-04T01:12:35.579166Z","title":"Skillrl: Evolving agents via recursive skill-augmented reinforcement learning.arXiv preprint arXiv:2602.08234, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:35.579166Z"},"links":{"cited_paper":"/paper/2602.08234","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:0cadd347c2004c4430b66a711192e626d454cf1b1b37eac36bac7daf5ad32675","observation_id":"e6271b35-7585-4cff-ac8d-17152f2132e8","resolution":{"observed_at":"2026-08-04T01:12:35.579166Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.06614","last_updated":"2026-05-07T17:31:50Z","snapshot_observed_at":"2026-07-06T23:19:05.764765Z","submitted_at":"2026-05-07T17:31:50Z","title":"SkillOS: Learning Skill Curation for Self-Evolving Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.06614","snapshot_observed_at":"2026-08-04T01:12:35.690973Z","title":"Skillos: Learning skill curation for self-evolving agents.arXiv preprint arXiv:2605.06614, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:35.690973Z"},"links":{"cited_paper":"/paper/2605.06614","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:4f7f60a11608bab86d5d01f78cd9848efafd18b325e0214ce3c8ec4c4c957420","observation_id":"1fcce961-c4c5-4a04-8cba-613fc43b0b43","resolution":{"observed_at":"2026-08-04T01:12:35.690973Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2603.02766","last_updated":"2026-03-03T09:07:22Z","snapshot_observed_at":"2026-08-04T19:32:12.955430Z","submitted_at":"2026-03-03T09:07:22Z","title":"EvoSkill: Automated Skill Discovery for Multi-Agent Systems","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2603.02766","snapshot_observed_at":"2026-08-04T01:12:35.780877Z","title":"Evoskill: Automated skill discovery for multi-agent systems.arXiv preprint arXiv:2603.02766, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:35.780877Z"},"links":{"cited_paper":"/paper/2603.02766","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:990287ecfc8236bca33d50f7ef015f871af76ade3f30afe399f68df1465ee229","observation_id":"60d83db8-2aa5-4635-b235-d2f7d73e311e","resolution":{"observed_at":"2026-08-04T01:12:35.780877Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2606.06741","last_updated":"2026-06-04T21:55:48Z","snapshot_observed_at":"2026-07-06T23:46:28.942186Z","submitted_at":"2026-06-04T21:55:48Z","title":"OpenSkill: Open-World Self-Evolution for LLM Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2606.06741","snapshot_observed_at":"2026-08-04T01:12:35.900154Z","title":"Yu, Ran Xu, Xiang Li, and Lichao Sun","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:35.900154Z"},"links":{"cited_paper":"/paper/2606.06741","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:255dca5a0bfb80d4f2960399d0a55baf44e99892cbf8945e5ea3c143070277c8","observation_id":"77a4cd23-7cf0-42d9-bfbc-c364071b1d2d","resolution":{"observed_at":"2026-08-04T01:12:35.900154Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.23904","last_updated":"2026-05-25T17:58:16Z","snapshot_observed_at":"2026-08-05T06:54:08.846880Z","submitted_at":"2026-05-22T17:59:50Z","title":"SkillOpt: Executive Strategy for Self-Evolving Agent Skills","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.23904","snapshot_observed_at":"2026-08-04T01:12:36.013114Z","title":"Skillopt: Executive strategy for self-evolving agent skills.arXiv preprint arXiv:2605.23904, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:36.013114Z"},"links":{"cited_paper":"/paper/2605.23904","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:acc955256b997f54efdef7fee39fb27a17409a86b3db1ba8aaa83b50b37e627e","observation_id":"3406787a-4e71-4b33-9ab7-b8a54d9deec6","resolution":{"observed_at":"2026-08-04T01:12:36.013114Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2603.28052","last_updated":"2026-03-30T05:33:50Z","snapshot_observed_at":"2026-08-07T21:46:16.930745Z","submitted_at":"2026-03-30T05:33:50Z","title":"Meta-Harness: End-to-End Optimization of Model Harnesses","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2603.28052","snapshot_observed_at":"2026-08-04T01:12:36.133344Z","title":"Meta-harness: End-to-end optimization of model harnesses.arXiv preprint arXiv:2603.28052, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:36.133344Z"},"links":{"cited_paper":"/paper/2603.28052","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:014ec49c6fb0bebc52ab2aff8e69bb4c1deda0e1d961c27b5a253815dacabdb7","observation_id":"9f8206b5-8e87-4d2a-82ea-3561f5a75200","resolution":{"observed_at":"2026-08-04T01:12:36.133344Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:36.247731Z","title":"Evoconfig: Self-evolving multi-agent systems for efficient autonomous environment configuration.arXiv preprint arXiv:2601.16489, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:36.247731Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:ddbc163a6d868f19fe1b24155cda1113c107c81c784e5f05ed954e6a038b2b74","observation_id":"f4deb2ae-ff8d-44e6-a47e-4bf4271f2963","resolution":{"observed_at":"2026-08-04T01:12:36.247731Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:36.328391Z","title":"Selaur: Self evolving llm agent via uncertainty-aware rewards","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:36.328391Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:b5d9a0b0387257c7db666cb84a93340162a9d259ec76da7db5667f4618b09f73","observation_id":"fd65faba-7aaf-4848-ae01-57b277ee676c","resolution":{"observed_at":"2026-08-04T01:12:36.328391Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.28814","last_updated":"2026-05-27T17:59:15Z","snapshot_observed_at":"2026-07-06T23:38:19.096349Z","submitted_at":"2026-05-27T17:59:15Z","title":"Self-Improving Language Models with Bidirectional Evolutionary Search","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.28814","snapshot_observed_at":"2026-08-04T01:12:36.464200Z","title":"Kakade, and Yilun Du","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:36.464200Z"},"links":{"cited_paper":"/paper/2605.28814","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:00d9220647bec3f30545688e1530304f2cd818a41ed73bdb32d4a1ae46945adc","observation_id":"78298233-2985-415e-a37a-9f4e5254b1ca","resolution":{"observed_at":"2026-08-04T01:12:36.464200Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:36.602393Z","title":"Tool-r0: Self-evolving llm agents for tool-learning from zero data.arXiv preprint arXiv:2602.21320, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:36.602393Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:d9a7ef248048a4a849419db6c4eeddbda887e16d87d37e23dd6e39005e2c5e91","observation_id":"42b75e4e-19e0-460c-86e8-ec6acc6fd764","resolution":{"observed_at":"2026-08-04T01:12:36.602393Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:36.737392Z","title":"Rein- forcing chain-of-thought reasoning with self-evolving rubrics.arXiv preprint arXiv:2602.10885, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:36.737392Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:f4521579b75f8d95ac936a135b10ea1cecd83b3bd9240100cf2fb6941f27945b","observation_id":"c2e9eb84-99d0-47dc-beda-b6654d619d6a","resolution":{"observed_at":"2026-08-04T01:12:36.737392Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:36.901016Z","title":"Metagen: Self-evolving roles and topologies for multi-agent llm reasoning.arXiv preprint arXiv:2601.19290, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:36.901016Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:08460f0ca27fcf513b335172bfcf43fe3a0bf566f4772ff8f897a9421127674d","observation_id":"8d4c3a23-e484-4c81-82a6-2206d90e5e79","resolution":{"observed_at":"2026-08-04T01:12:36.901016Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:37.006298Z","title":"Self-evolving multi-agent collaboration networks for software development","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:37.006298Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:55cee944677071d7514f68c04add6658f5a85c44189a89709f7cdc733ef0f982","observation_id":"81c2eb88-de5c-4ea8-b3e8-4ef04ac48b49","resolution":{"observed_at":"2026-08-04T01:12:37.006298Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.18646","last_updated":"2026-04-14T11:23:32Z","snapshot_observed_at":"2026-08-02T05:01:07.687054Z","submitted_at":"2025-05-24T11:12:14Z","title":"SEW: Self-Evolving Agentic Workflows for Automated Code Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.18646","snapshot_observed_at":"2026-08-04T01:12:37.085301Z","title":"Sew: Self-evolving agentic workflows for automated code generation.arXiv preprint arXiv:2505.18646, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:37.085301Z"},"links":{"cited_paper":"/paper/2505.18646","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:65f5f011a1e95cf4220b98c00b93807b78bb6854cb772d1fdc6d62ac66cbdb2c","observation_id":"fb119d41-1db2-491c-b863-2c7cd2db0aad","resolution":{"observed_at":"2026-08-04T01:12:37.085301Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:37.179134Z","title":"Evotool: Self-evolving tool-use policy optimization in llm agents via blame-aware mutation and diversity-aware selection.arXiv preprint arXiv:2603.04900, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:37.179134Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:32b0b699266db833003fa2406babde4d2ab232342def20301ed1039c51c08f8f","observation_id":"a438e6fe-bc99-4020-aadc-cd4a1d5f5693","resolution":{"observed_at":"2026-08-04T01:12:37.179134Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13131","last_updated":"2025-06-16T06:37:18Z","snapshot_observed_at":"2026-08-07T04:21:43.190472Z","submitted_at":"2025-06-16T06:37:18Z","title":"AlphaEvolve: A coding agent for scientific and algorithmic discovery","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.13131","snapshot_observed_at":"2026-08-04T01:12:37.248349Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:37.248349Z"},"links":{"cited_paper":"/paper/2506.13131","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:8191519f3616a305becff104dd6650ccf81f4c38bbf01b09ecdeb6abbd632c95","observation_id":"3ecd6a5f-8ebc-4011-8b8c-1248324a388d","resolution":{"observed_at":"2026-08-04T01:12:37.248349Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:37.318781Z","title":"Evotest: Evolutionary test-time learning for self-improving agentic systems","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:37.318781Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:3a322347f0393b86762444fe4f3d8a14404e43aed66fd39df2d1323c5e49bf4c","observation_id":"6a19d7cd-4a82-425a-8a93-a521749a1200","resolution":{"observed_at":"2026-08-04T01:12:37.318781Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:37.425048Z","title":"Building self-evolving agents via experience-driven lifelong learning: A framework and benchmark.arXiv preprint arXiv:2508.19005, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:37.425048Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:9c2dae669b9b5119372ae0b31c7c0aebf03d0b19a2302f72818ca01a4a4d7e33","observation_id":"3a3b0b20-44fa-43b8-a733-cfb5ed07a0be","resolution":{"observed_at":"2026-08-04T01:12:37.425048Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:37.498006Z","title":"Optimizing generative ai by backpropagating language model feedback.Nature, 639:609–616, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:37.498006Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:8cb6c539f252775b429d428a7d2727d6a62e6b5e37a803cfd215904bd326a582","observation_id":"b76f30c6-aa60-4f0a-8914-2712ff2475be","resolution":{"observed_at":"2026-08-04T01:12:37.498006Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:37.591245Z","title":"Your agent may misevolve: Emergent risks in self-evolving llm agents","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:37.591245Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:58226d92f5d6cb0a80956d55e329bd381bec6403b38bf4c913fbaf0b93298e42","observation_id":"73fce81a-b1bf-4152-9555-ee4a82465c9c","resolution":{"observed_at":"2026-08-04T01:12:37.591245Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:37.698493Z","title":"Webshop: Towards scalable real-world web interaction with grounded language agents","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:37.698493Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:054a2b0e3ca740dc20ae139585370b2cf09032ac41fba6834d950db067775b4b","observation_id":"d6b8e116-6b94-4a8e-8c9e-eafca8f43605","resolution":{"observed_at":"2026-08-04T01:12:37.698493Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:37.811240Z","title":"Xu, Hao Zhu, Xuhui Zhou, Robert Lo, Abishek Sridhar, Xianyi Cheng, Tianyue Ou, Yonatan Bisk, Daniel Fried, Uri Alon, and Graham Neubig","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:37.811240Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:7d3f23c74cb256cfbe6b78c526c7630b33c043ee5ad94a9705e7657133186825","observation_id":"2935ea0f-764a-4a2a-8341-5d42cd473aad","resolution":{"observed_at":"2026-08-04T01:12:37.811240Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:37.916387Z","title":"Webvoyager: Building an end-to-end web agent with large multimodal models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:37.916387Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:4aeb4f1ffdb5bd1eb26a5626d5df90c615d1c592f03d4887f6e19849cf0d21d1","observation_id":"188f85ae-edff-4a3f-aa09-91fc9df16cc3","resolution":{"observed_at":"2026-08-04T01:12:37.916387Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:37.996496Z","title":"Osworld: Benchmarking multimodal agents for open-ended tasks in real computer environments","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:37.996496Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:1282f63a041f77a44220da7c3cf8982ff492dd99312274e479af63e29951fdfe","observation_id":"bf78fd0c-0c32-48c7-86ad-fc0033c497ef","resolution":{"observed_at":"2026-08-04T01:12:37.996496Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.00933","last_updated":"2026-05-19T23:26:22Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-01-31T23:19:39Z","title":"MCP-Atlas: A Large-Scale Benchmark for Tool-Use Competency with Real MCP Servers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.00933","snapshot_observed_at":"2026-08-04T01:12:38.075464Z","title":"Mcp-atlas: A large-scale benchmark for tool-use competency with real mcp servers.arXiv preprint arXiv:2602.00933, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:38.075464Z"},"links":{"cited_paper":"/paper/2602.00933","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:cfb5c88e5b4f357b751f71c8326536e78baa5e0217c2fc6276e38fca48b70f7c","observation_id":"592bab4f-14f5-43de-8234-c160ebd7d284","resolution":{"observed_at":"2026-08-04T01:12:38.075464Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:38.158997Z","title":"The tool decathlon: Benchmarking language agents for diverse, realistic, and long-horizon task execution","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:38.158997Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:87a4200b2168ce532f49dd3d5e4cfe555b96fa3b829194ff79952211305317f9","observation_id":"f05cad0a-6ce5-48e5-82e9-5d21a17a8a02","resolution":{"observed_at":"2026-08-04T01:12:38.158997Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:38.214251Z","title":"Terminal-bench: Benchmarking agents on hard, realistic tasks in command line interfaces","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:38.214251Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:a7cd5f2325bf8072f1f16a2a496fa5c4719ae9b184565425b6160f67cee35cbc","observation_id":"02c2e747-2463-4d6e-b47b-e9c0e5f07032","resolution":{"observed_at":"2026-08-04T01:12:38.214251Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:38.287770Z","title":"Cybergym: Evaluating AI agents’ real-world cybersecurity capabilities at scale","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:38.287770Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:6a88bfcfc434b398302d74d7b7b5f6caaab72155cb637578768e266252fe482b","observation_id":"4cc7197c-1adf-4548-8a30-f846e95381f2","resolution":{"observed_at":"2026-08-04T01:12:38.287770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.12670","last_updated":"2026-03-13T07:33:01Z","snapshot_observed_at":"2026-07-06T22:45:44.135338Z","submitted_at":"2026-02-13T07:06:06Z","title":"SkillsBench: Benchmarking How Well Agent Skills Work Across Diverse Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.12670","snapshot_observed_at":"2026-08-04T01:12:38.372036Z","title":"Skillsbench: Benchmarking how well agent skills work across diverse tasks.arXiv preprint arXiv:2602.12670, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:38.372036Z"},"links":{"cited_paper":"/paper/2602.12670","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:e51e34edf3d1397773dd323b93b3d657205b1714888b523af65458b10a93c43d","observation_id":"81771bf5-edef-4fed-94ad-808de9210a98","resolution":{"observed_at":"2026-08-04T01:12:38.372036Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.17308","last_updated":"2026-04-19T07:51:46Z","snapshot_observed_at":"2026-07-06T23:04:28.260463Z","submitted_at":"2026-04-19T07:51:46Z","title":"SkillFlow:Benchmarking Lifelong Skill Discovery and Evolution for Autonomous Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.17308","snapshot_observed_at":"2026-08-04T01:12:38.455621Z","title":"Skillflow: Benchmarking lifelong skill discovery and evolution for autonomous agents.arXiv preprint arXiv:2604.17308, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:38.455621Z"},"links":{"cited_paper":"/paper/2604.17308","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:746941a071f24176e4ad3868375b47a295e6fe39df8a1da34ac995ef4472c14c","observation_id":"b6e25965-c2c8-4b22-b36c-349ab6a533c4","resolution":{"observed_at":"2026-08-04T01:12:38.455621Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:38.536812Z","title":"Gaia: a benchmark for general ai assistants","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:38.536812Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:7cff75eecdedc4e3f88bc10379803474cf29211753de7c93b43fe2f71ee3c1d3","observation_id":"b47a9d7b-74e2-4fe3-8fbd-963edfbf6074","resolution":{"observed_at":"2026-08-04T01:12:38.536812Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.12516","last_updated":"2025-04-16T22:27:45Z","snapshot_observed_at":"2026-08-03T00:43:33.338074Z","submitted_at":"2025-04-16T22:27:45Z","title":"BrowseComp: A Simple Yet Challenging Benchmark for Browsing Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.12516","snapshot_observed_at":"2026-08-04T01:12:38.625544Z","title":"Browsecomp: A simple yet challenging benchmark for browsing agents.arXiv preprint arXiv:2504.12516, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:38.625544Z"},"links":{"cited_paper":"/paper/2504.12516","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:536cf64996f23551adbb2a39709399b96e68611bf3bd1f1db6495be91a304507","observation_id":"b512849c-9191-4ddc-a38b-e8badd0f1281","resolution":{"observed_at":"2026-08-04T01:12:38.625544Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2510.04374","last_updated":"2025-10-05T21:36:43Z","snapshot_observed_at":"2026-07-06T22:31:44.964774Z","submitted_at":"2025-10-05T21:36:43Z","title":"GDPval: Evaluating AI Model Performance on Real-World Economically Valuable Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2510.04374","snapshot_observed_at":"2026-08-04T01:12:38.705418Z","title":"Gdpval: Evaluating ai model performance on real-world economically valuable tasks.arXiv preprint arXiv:2510.04374, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:38.705418Z"},"links":{"cited_paper":"/paper/2510.04374","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:a13721b28f2f1bdd330538a396dc07a4904226e463b21c1af547fec625e46c9d","observation_id":"6f0dc2de-6651-456f-a1aa-1c902b2c9a9f","resolution":{"observed_at":"2026-08-04T01:12:38.705418Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-04T01:12:38.779498Z","title":"Finance agent benchmark: Benchmarking llms on real-world financial research tasks.arXiv preprint arXiv:2508.00828, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:38.779498Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:7ff23ca049e643e6a7e73272a64caf210b67a894bf6d5e61be6a2622a9d34a8d","observation_id":"9f015896-0a35-463c-a30a-23394db146bc","resolution":{"observed_at":"2026-08-04T01:12:38.779498Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:38.857487Z","title":"Agent workflow memory","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:38.857487Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:f081e044a388383b6fda71dbb4589f01df1ec3afb64500ac584f2876e80d08a6","observation_id":"ba0c0772-e8a4-4962-8322-b8fd75fcc637","resolution":{"observed_at":"2026-08-04T01:12:38.857487Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:38.939920Z","title":"none identified","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:38.939920Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:8fc1283a3d25332381b8c5a86553899739a1aa2ce35e40aca0e4c853c9700a1a","observation_id":"62aa1b8c-60ad-45aa-9b9b-2d5cc21f14e3","resolution":{"observed_at":"2026-08-04T01:12:38.939920Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:39.018297Z","title":"reasoning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:39.018297Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:d7fda506f50f0c30b6069011a58ecf033042bf04bf8cf74d19291bb69db7ef42","observation_id":"3377049d-f4bc-4c76-8fb7-557fa03b9c9e","resolution":{"observed_at":"2026-08-04T01:12:39.018297Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:39.092541Z","title":"Order from most to least important","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:39.092541Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:2db56d166e65ba7da9631e3ac387cc8bb7482b135914119d1123a5395ecfe113","observation_id":"e2a5d1a1-0e80-419f-9eef-e66ae93408dc","resolution":{"observed_at":"2026-08-04T01:12:39.092541Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:39.153063Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:39.153063Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:0d4b2672f8d1cd0c4bd16e36325b9778bbf9527d514b5f7bcbb6c9e514933069","observation_id":"f10004a5-26a9-4d39-95d1-277af5e60038","resolution":{"observed_at":"2026-08-04T01:12:39.153063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:39.230523Z","title":"Status:","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:39.230523Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:38f49287fb973717305fac9f42a2edbd32653fbad92697a95b12241c415c32c6","observation_id":"90dc9e6e-7d6c-4c87-8624-ff0085958f7c","resolution":{"observed_at":"2026-08-04T01:12:39.230523Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:39.310212Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:39.310212Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:e6de1a8decefb80212ac8d310bef1e6709dadbd1b5e2d58f1e261dab5c168b9e","observation_id":"53aba8b1-c42c-45e0-9c21-a9a2afb989b4","resolution":{"observed_at":"2026-08-04T01:12:39.310212Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:39.419533Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:39.419533Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:ec46fe8c39517987731f01fd1e6ca3d8e0a756f9c3c20ba2f943b5b7896006d1","observation_id":"09cd7c8b-48af-4bba-ade3-477b10cbe395","resolution":{"observed_at":"2026-08-04T01:12:39.419533Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:39.476352Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:39.476352Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:8a2546434734ecd31931d9fdfa7dfc9862487cb75470192409a63f315d22a46b","observation_id":"00a09f38-aa8c-413d-a4b4-396c781e0e26","resolution":{"observed_at":"2026-08-04T01:12:39.476352Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:39.559055Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:39.559055Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:443a4fbc81f358f46f37e0580b97e7338e5192f472daa62b70888b49b3d17447","observation_id":"6984840c-21ed-47f3-a1db-0713b084276d","resolution":{"observed_at":"2026-08-04T01:12:39.559055Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":99,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":102},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 100 of 102 outbound references and 0 inbound Pith citation observations for arXiv:2608.00155."}