{"as_of":"2026-08-10T04:21:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:87802c009a48f8a9cf77daf7a408a84a311d08b4a1586100c48d51bd21770870","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":32,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":32,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":32,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":32,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T22:11:21.458007Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":"2508.00828","doi":"10.48550/arxiv.2508.00828","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Finance agent benchmark: Benchmarking llms on real-world financial research tasks","venue":"ArXiv.org","work_id":"7b3f7b73-fcec-452b-a615-268a46e2938f","year":2025},"citing_paper":{"arxiv_id":"2512.13168","last_updated":"2026-04-15T17:46:00Z","snapshot_observed_at":"2026-08-08T14:46:54.054418Z","submitted_at":"2025-12-15T10:28:45Z","title":"Finch: Benchmarking Finance & Accounting across Spreadsheet-Centric Enterprise Workflows","version":5},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-16T22:43:48.618334Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2512.13168"},"observation_digest":"sha256:61615f5717beaa6f97e038f364cf860848576f662f226697b1f34c80fc40268a","observation_id":"e3fb759d-a9ee-40c1-b8d5-49ca76f4aaf5","resolution":{"observed_at":"2026-05-16T22:48:38.318919Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-04T05:54:53.348743Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.08262","last_updated":"2026-08-01T16:26:47Z","snapshot_observed_at":"2026-08-08T02:40:15.172458Z","submitted_at":"2026-03-09T11:33:05Z","title":"FinToolBench: Evaluating LLM Agents for Real-World Financial Tool Use","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-04T05:54:53.348743Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2603.08262"},"observation_digest":"sha256:aac8acfd12943ec72845d341c63bf9276ba2f07c6941dca4e6c7e9dba38a9714","observation_id":"9884bc28-e204-49d0-b0b1-909bc64e538c","resolution":{"observed_at":"2026-08-04T05:54:53.348743Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":"2508.00828","doi":"10.48550/arxiv.2508.00828","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Finance agent benchmark: Benchmarking llms on real-world financial research tasks","venue":"ArXiv.org","work_id":"7b3f7b73-fcec-452b-a615-268a46e2938f","year":2025},"citing_paper":{"arxiv_id":"2604.11304","last_updated":"2026-04-13T11:02:32Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-13T11:02:32Z","title":"BankerToolBench: Evaluating AI Agents in End-to-End Investment Banking Workflows","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T15:55:49.453068Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2604.11304"},"observation_digest":"sha256:8f2a42f05b2b7452a1ecb4a2e509ee656b884f58807bc644607e01507bf04081","observation_id":"6284342d-4f24-46de-b654-f13aa26fe2ae","resolution":{"observed_at":"2026-05-11T09:36:03.144299Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":"2508.00828","doi":"10.48550/arxiv.2508.00828","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Finance agent benchmark: Benchmarking llms on real-world financial research tasks","venue":"ArXiv.org","work_id":"7b3f7b73-fcec-452b-a615-268a46e2938f","year":2025},"citing_paper":{"arxiv_id":"2604.17305","last_updated":"2026-04-19T07:42:07Z","snapshot_observed_at":"2026-08-03T00:49:45.756423Z","submitted_at":"2026-04-19T07:42:07Z","title":"BizCompass: Benchmarking the Reasoning Capabilities of LLMs in Business Knowledge and Applications","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T05:52:40.026883Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2604.17305"},"observation_digest":"sha256:0f8d043474d24ff26766cad29ddfd3268fd5afdeaa5bd937ec61e488fb11ade6","observation_id":"e171615f-3b16-499a-baad-b554af3ff0d7","resolution":{"observed_at":"2026-05-10T05:56:11.236886Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":"2508.00828","doi":"10.48550/arxiv.2508.00828","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Finance agent benchmark: Benchmarking llms on real-world financial research tasks","venue":"ArXiv.org","work_id":"7b3f7b73-fcec-452b-a615-268a46e2938f","year":2025},"citing_paper":{"arxiv_id":"2604.22820","last_updated":"2026-04-17T15:31:20Z","snapshot_observed_at":"2026-07-06T23:09:10.050398Z","submitted_at":"2026-04-17T15:31:20Z","title":"Complete Cyclic Subtask Graphs for Tool-Using LLM Agents: Flexibility, Cost, and Bottlenecks in Multi-Agent Workflows","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-10T07:05:43.392997Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2604.22820"},"observation_digest":"sha256:1b403ce56e8b44914a06dc3a6f5ec7f42a3c35227a7025b47b0b0e8fe67de90b","observation_id":"999dfa21-90aa-4b40-b381-cefcf3845150","resolution":{"observed_at":"2026-05-10T07:06:52.862114Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":"2508.00828","doi":"10.48550/arxiv.2508.00828","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Finance agent benchmark: Benchmarking llms on real-world financial research tasks","venue":"ArXiv.org","work_id":"7b3f7b73-fcec-452b-a615-268a46e2938f","year":2025},"citing_paper":{"arxiv_id":"2604.24668","last_updated":"2026-06-09T15:03:07Z","snapshot_observed_at":"2026-07-06T23:10:38.548416Z","submitted_at":"2026-04-27T16:27:10Z","title":"The Price of Agreement: Measuring LLM Sycophancy in Agentic Financial Applications","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-08T03:22:45.216764Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2604.24668"},"observation_digest":"sha256:7217d77d469ad60f4e66d717de688d4358918154f8ea5f6e9ba88e47f3a1c5d1","observation_id":"b7e87a0c-cc3c-436e-987f-328675b5df4f","resolution":{"observed_at":"2026-05-11T22:06:25.659426Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":"2508.00828","doi":"10.48550/arxiv.2508.00828","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Finance agent benchmark: Benchmarking llms on real-world financial research tasks","venue":"ArXiv.org","work_id":"7b3f7b73-fcec-452b-a615-268a46e2938f","year":2025},"citing_paper":{"arxiv_id":"2604.26235","last_updated":"2026-04-29T02:32:14Z","snapshot_observed_at":"2026-08-02T11:16:41.718755Z","submitted_at":"2026-04-29T02:32:14Z","title":"LATTICE: Evaluating Decision Support Utility of Crypto Agents","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-07T13:30:46.523784Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2604.26235"},"observation_digest":"sha256:07e9ff980a23b008609b1103adb68ebf05649b43815a7c5a13f8953098e50b90","observation_id":"96a8db9d-9314-4862-88a8-186b0d4afc79","resolution":{"observed_at":"2026-05-12T08:56:25.160988Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":"2508.00828","doi":"10.48550/arxiv.2508.00828","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Finance agent benchmark: Benchmarking llms on real-world financial research tasks","venue":"ArXiv.org","work_id":"7b3f7b73-fcec-452b-a615-268a46e2938f","year":2025},"citing_paper":{"arxiv_id":"2605.09539","last_updated":"2026-05-10T13:52:00Z","snapshot_observed_at":"2026-08-06T17:57:07.648045Z","submitted_at":"2026-05-10T13:52:00Z","title":"TacoMAS: Test-Time Co-Evolution of Topology and Capability in LLM-based Multi-Agent Systems","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-05-12T04:53:54.754878Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2605.09539"},"observation_digest":"sha256:baad28fe0c25f48d20ccb8c4ff65dc234cd3557d1aaaa98c69bb79ca5032ca41","observation_id":"66724bfa-d850-4f92-9cef-f5d02839c0d4","resolution":{"observed_at":"2026-05-12T05:51:24.466508Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":"2508.00828","doi":"10.48550/arxiv.2508.00828","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Finance agent benchmark: Benchmarking llms on real-world financial research tasks","venue":"ArXiv.org","work_id":"7b3f7b73-fcec-452b-a615-268a46e2938f","year":2025},"citing_paper":{"arxiv_id":"2605.14355","last_updated":"2026-05-14T04:30:49Z","snapshot_observed_at":"2026-08-03T03:55:27.049390Z","submitted_at":"2026-05-14T04:30:49Z","title":"Herculean: An Agentic Benchmark for Financial Intelligence","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-30T21:06:46.156943Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2605.14355"},"observation_digest":"sha256:ff882cc03ec1967ae619c31d22936397f8fd0744d222d712183a7a8e26ecc0a8","observation_id":"e65918f3-0ecb-430b-b823-d9ebea59eb22","resolution":{"observed_at":"2026-06-30T21:15:04.687390Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":"2508.00828","doi":"10.48550/arxiv.2508.00828","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Finance agent benchmark: Benchmarking llms on real-world financial research tasks","venue":"ArXiv.org","work_id":"7b3f7b73-fcec-452b-a615-268a46e2938f","year":2025},"citing_paper":{"arxiv_id":"2605.27466","last_updated":"2026-05-26T08:10:52Z","snapshot_observed_at":"2026-08-07T04:29:44.927652Z","submitted_at":"2026-05-26T08:10:52Z","title":"AgensFlow: A Coordination-Policy Substrate for Multi-Agent Systems","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-01T16:08:56.103382Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2605.27466"},"observation_digest":"sha256:89b7c3541437f50122b437c0223e27326acea1119eef3df533aa8ad776879df5","observation_id":"6ceb1d54-aa04-4b95-b713-35fb9571c04a","resolution":{"observed_at":"2026-07-01T16:15:49.116976Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":"2508.00828","doi":"10.48550/arxiv.2508.00828","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Finance agent benchmark: Benchmarking llms on real-world financial research tasks","venue":"ArXiv.org","work_id":"7b3f7b73-fcec-452b-a615-268a46e2938f","year":2025},"citing_paper":{"arxiv_id":"2606.00939","last_updated":"2026-05-31T00:53:05Z","snapshot_observed_at":"2026-07-06T23:41:34.459700Z","submitted_at":"2026-05-31T00:53:05Z","title":"FinCom: A Financial Multi-Agent Demo with Disagree-or-Commit Deliberation","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-28T16:36:33.579549Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2606.00939"},"observation_digest":"sha256:19f545941003793258df0743711c3634ce1260ec509940d48f2bc29cd9aa6a7b","observation_id":"45097e83-8b06-49ae-82ef-ab6f1d3c7248","resolution":{"observed_at":"2026-07-01T21:36:15.568412Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":"2508.00828","doi":"10.48550/arxiv.2508.00828","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Finance agent benchmark: Benchmarking llms on real-world financial research tasks","venue":"ArXiv.org","work_id":"7b3f7b73-fcec-452b-a615-268a46e2938f","year":2025},"citing_paper":{"arxiv_id":"2606.01886","last_updated":"2026-06-01T08:31:35Z","snapshot_observed_at":"2026-08-05T19:42:50.870602Z","submitted_at":"2026-06-01T08:31:35Z","title":"Absorbing Complexity: An Interaction-Native Knowledge Harness for Financial LLM Agents","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-28T14:57:00.865047Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2606.01886"},"observation_digest":"sha256:c886b8e3225df3cb0d7c241b01d56ee6e0c1c0831d301ba0fe956b053234ffa1","observation_id":"ff2b5e76-4b6d-4462-a8f4-53e185784f01","resolution":{"observed_at":"2026-07-01T22:56:19.783133Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":"2508.00828","doi":"10.48550/arxiv.2508.00828","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Finance agent benchmark: Benchmarking llms on real-world financial research tasks","venue":"ArXiv.org","work_id":"7b3f7b73-fcec-452b-a615-268a46e2938f","year":2025},"citing_paper":{"arxiv_id":"2606.02859","last_updated":"2026-06-01T20:21:09Z","snapshot_observed_at":"2026-07-29T22:03:03.516670Z","submitted_at":"2026-06-01T20:21:09Z","title":"Economy of Minds: Emerging Multi-Agent Intelligence with Economic Interactions","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-28T14:26:50.444070Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2606.02859"},"observation_digest":"sha256:e64f4499e5d228f7234f7169862e5ed858870fc4d20e09cde1e4dc02b66d82f7","observation_id":"431cc5cb-40b3-4d56-bbb2-8c087d4fe592","resolution":{"observed_at":"2026-07-01T23:26:21.796442Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":"2508.00828","doi":"10.48550/arxiv.2508.00828","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Finance agent benchmark: Benchmarking llms on real-world financial research tasks","venue":"ArXiv.org","work_id":"7b3f7b73-fcec-452b-a615-268a46e2938f","year":2025},"citing_paper":{"arxiv_id":"2606.03829","last_updated":"2026-06-02T16:12:34Z","snapshot_observed_at":"2026-08-06T15:50:02.772309Z","submitted_at":"2026-06-02T16:12:34Z","title":"BigFinanceBench: A Workflow-Grounded Benchmark for Financial-Research Agents","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-28T09:35:28.694416Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2606.03829"},"observation_digest":"sha256:631e43362608ecd8d0df8cd4bc3a6d2afb8f11665a7fb8124886d7b46d9d891f","observation_id":"d787c9c0-45ef-42d1-887b-6eae9a6afb40","resolution":{"observed_at":"2026-07-02T03:56:34.697572Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":"2508.00828","doi":"10.48550/arxiv.2508.00828","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Finance agent benchmark: Benchmarking llms on real-world financial research tasks","venue":"ArXiv.org","work_id":"7b3f7b73-fcec-452b-a615-268a46e2938f","year":2025},"citing_paper":{"arxiv_id":"2606.05661","last_updated":"2026-06-04T03:43:28Z","snapshot_observed_at":"2026-08-09T08:46:24.991941Z","submitted_at":"2026-06-04T03:43:28Z","title":"Continual Learning Bench: Evaluating Frontier AI Systems in Real-World Stateful Environments","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-28T01:45:28.693098Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2606.05661"},"observation_digest":"sha256:322d25f69e16449ce1be78db8cf3cd7d68c7e114fabbb530fd3bcd39c5afd16d","observation_id":"79b3bbbc-9048-441d-b92a-6d3c79891d38","resolution":{"observed_at":"2026-07-02T12:56:56.868237Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":"2508.00828","doi":"10.48550/arxiv.2508.00828","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Finance agent benchmark: Benchmarking llms on real-world financial research tasks","venue":"ArXiv.org","work_id":"7b3f7b73-fcec-452b-a615-268a46e2938f","year":2025},"citing_paper":{"arxiv_id":"2606.11166","last_updated":"2026-06-09T17:46:10Z","snapshot_observed_at":"2026-08-07T21:17:06.825949Z","submitted_at":"2026-06-09T17:46:10Z","title":"Flaws in the LLM Automation Narrative","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-06-27T10:52:36.252919Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2606.11166"},"observation_digest":"sha256:6186e658755b5a1d47dbe24f2915a550c559d09baca82df71d835d4fe692ded4","observation_id":"83ed772e-a5ac-4a60-9eb7-446dda01e3ab","resolution":{"observed_at":"2026-07-03T08:17:45.296220Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":"2508.00828","doi":"10.48550/arxiv.2508.00828","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Finance agent benchmark: Benchmarking llms on real-world financial research tasks","venue":"ArXiv.org","work_id":"7b3f7b73-fcec-452b-a615-268a46e2938f","year":2025},"citing_paper":{"arxiv_id":"2606.12191","last_updated":"2026-06-10T15:15:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-06-10T15:15:01Z","title":"Agentic Environment Engineering for Large Language Models: A Survey of Environment Modeling, Synthesis, Evaluation, and Application","version":1},"reference_index":161,"source":"pdf_text","source_observed_at":"2026-06-27T09:46:30.702256Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2606.12191"},"observation_digest":"sha256:38b7ff2db428d49d0c58009791a0a3dc7d5d4c77f22a6f3fa196ddef55fef740","observation_id":"27832513-4dfe-43ab-8c5e-5bcde2e0dead","resolution":{"observed_at":"2026-06-27T09:50:48.360204Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":"2508.00828","doi":"10.48550/arxiv.2508.00828","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Finance agent benchmark: Benchmarking llms on real-world financial research tasks","venue":"ArXiv.org","work_id":"7b3f7b73-fcec-452b-a615-268a46e2938f","year":2025},"citing_paper":{"arxiv_id":"2606.13608","last_updated":"2026-06-11T17:23:54Z","snapshot_observed_at":"2026-08-07T11:12:52.411120Z","submitted_at":"2026-06-11T17:23:54Z","title":"AgentBeats: Agentifying Agent Assessment for Openness, Standardization, and Reproducibility","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-27T06:41:41.799596Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2606.13608"},"observation_digest":"sha256:6f75fb210663c38b2c880b32f6cb917174971e54d623b10de28f6a75a9d45b3f","observation_id":"21d6a224-e484-4044-bf87-3486d8f51bb1","resolution":{"observed_at":"2026-07-03T15:08:33.127177Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":"2508.00828","doi":"10.48550/arxiv.2508.00828","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Finance agent benchmark: Benchmarking llms on real-world financial research tasks","venue":"ArXiv.org","work_id":"7b3f7b73-fcec-452b-a615-268a46e2938f","year":2025},"citing_paper":{"arxiv_id":"2606.22719","last_updated":"2026-06-21T23:36:04Z","snapshot_observed_at":"2026-08-08T14:32:12.215687Z","submitted_at":"2026-06-21T23:36:04Z","title":"Leakage-Aware Benchmarking of LLM Forecasting: Real-Time Nowcasts as the Decision-Time Input for Macro Factor Ranking","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-06-26T09:14:31.166883Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2606.22719"},"observation_digest":"sha256:43544b7ee4465d217a9ee35d3873471bd33cec29df665effa7ea24f95e33aa07","observation_id":"44df256c-fa33-45eb-9c09-2057f1d2f58c","resolution":{"observed_at":"2026-07-04T09:59:45.685029Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":"2508.00828","doi":"10.48550/arxiv.2508.00828","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Finance agent benchmark: Benchmarking llms on real-world financial research tasks","venue":"ArXiv.org","work_id":"7b3f7b73-fcec-452b-a615-268a46e2938f","year":2025},"citing_paper":{"arxiv_id":"2606.23032","last_updated":"2026-06-30T10:07:05Z","snapshot_observed_at":"2026-08-06T04:55:43.930167Z","submitted_at":"2026-06-22T08:42:19Z","title":"IPO Finance Agent: Benchmark of LLM Financial Analysts Beyond Finance Agent v2, with Automated Rubric Generation, on the SpaceX (SPCX) IPO","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-26T08:51:42.837201Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2606.23032"},"observation_digest":"sha256:93fdb1a62e868df26d6d7df927b49cbe7b16908cd3b25355c437b67ca40280a0","observation_id":"2d8e58db-f1c9-4ab0-9afc-aff113de5772","resolution":{"observed_at":"2026-07-04T10:29:44.669903Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":"2508.00828","doi":"10.48550/arxiv.2508.00828","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Finance agent benchmark: Benchmarking llms on real-world financial research tasks","venue":"ArXiv.org","work_id":"7b3f7b73-fcec-452b-a615-268a46e2938f","year":2025},"citing_paper":{"arxiv_id":"2606.23032","last_updated":"2026-06-30T10:07:05Z","snapshot_observed_at":"2026-08-06T04:55:43.930167Z","submitted_at":"2026-06-22T08:42:19Z","title":"IPO Finance Agent: Benchmark of LLM Financial Analysts Beyond Finance Agent v2, with Automated Rubric Generation, on the SpaceX (SPCX) IPO","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-01T06:49:27.958422Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2606.23032"},"observation_digest":"sha256:457e5dc0d3ff016c333967d185bcd60d48c1a8478b7a89cc8a54613f48c8ae97","observation_id":"2ed051a4-e875-47b9-b186-5aafe2fc6ebf","resolution":{"observed_at":"2026-07-01T08:55:35.925456Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":"2508.00828","doi":"10.48550/arxiv.2508.00828","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Finance agent benchmark: Benchmarking llms on real-world financial research tasks","venue":"ArXiv.org","work_id":"7b3f7b73-fcec-452b-a615-268a46e2938f","year":2025},"citing_paper":{"arxiv_id":"2606.26346","last_updated":"2026-06-24T19:38:21Z","snapshot_observed_at":"2026-08-08T08:43:50.725073Z","submitted_at":"2026-06-24T19:38:21Z","title":"How Do Tool-Augmented LLM Agents Perform on Real-World Energy Analytics Tasks?","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-26T01:34:10.103638Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2606.26346"},"observation_digest":"sha256:e480af605d08d286fe87af85ca0f61551d9ceab924af0cea5579d4203f2633c2","observation_id":"d041e6c9-afbe-4480-93b2-cd9adaf93e0c","resolution":{"observed_at":"2026-07-04T15:29:56.712681Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-01T23:29:50.743589Z","title":"Finance agent benchmark: A human-in-the-loop evaluation harness for LLM agents in the finance domain,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.15414","last_updated":"2026-07-16T19:38:23Z","snapshot_observed_at":"2026-08-07T21:26:09.000671Z","submitted_at":"2026-07-16T19:38:23Z","title":"AI Trading: Evaluating Large Language Models for Technical Market Analysis","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-01T23:29:50.743589Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2607.15414"},"observation_digest":"sha256:faf1584303260fcb8bc1575430360f267a6542708586f9d64e66508b0d939af8","observation_id":"fdf62176-6edf-42ed-872f-5b928de1a3a2","resolution":{"observed_at":"2026-08-01T23:29:50.743589Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-02T07:28:28.105727Z","title":"Finance agent benchmark: Benchmarking llms on real-world financial research tasks.arXiv preprint arXiv:2508.00828,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.19409","last_updated":"2026-07-11T01:16:31Z","snapshot_observed_at":"2026-08-09T17:37:34.254198Z","submitted_at":"2026-07-11T01:16:31Z","title":"FORCE-Bench: A Benchmark, Dataset, and Evaluation Harness for Agentic AI in Enterprise Finance","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-02T07:28:28.105727Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2607.19409"},"observation_digest":"sha256:393fcddbb6ba72414b72fd9e2696dd374afd1fd501c34a12e43e45e8a59d1bb8","observation_id":"61116e77-9e8f-44a3-84c3-56f372496f87","resolution":{"observed_at":"2026-08-02T07:28:28.105727Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-01T09:47:15.283030Z","title":"Finance agent benchmark: Benchmarking LLMs on real-world financial research tasks.arXiv preprint arXiv:2508.00828, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.20645","last_updated":"2026-07-22T18:15:12Z","snapshot_observed_at":"2026-08-09T00:36:51.624331Z","submitted_at":"2026-07-22T18:15:12Z","title":"Frontier Financial Judgement: Can agents tell what might move a stock?","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-01T09:47:15.283030Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2607.20645"},"observation_digest":"sha256:79217f0225c6a95c1e99ea101358cb66bfff2d082e05152fff33fd77f0f0adfb","observation_id":"77ef7915-b7a4-4dfe-92d0-0931651c2cb0","resolution":{"observed_at":"2026-08-01T09:47:15.283030Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-01T01:16:42.610725Z","title":"arXiv preprint arXiv:2508.00828 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.25891","last_updated":"2026-07-28T15:50:19Z","snapshot_observed_at":"2026-08-06T22:52:32.781663Z","submitted_at":"2026-07-28T15:50:19Z","title":"Messier: A High-Resolution Corpus for Cross-Benchmark Agent Evaluation","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-01T01:16:42.610725Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2607.25891"},"observation_digest":"sha256:ac674821949aae3977a99814b6e5cf49f8b45fd59284333c0bac0158f7963f71","observation_id":"7b9074c3-e3ab-4c76-a185-367336ed36bb","resolution":{"observed_at":"2026-08-01T01:16:42.610725Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-07-31T07:38:10.220810Z","title":"arXiv:2508.00828","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.28430","last_updated":"2026-07-30T16:07:32Z","snapshot_observed_at":"2026-08-06T21:05:53.103649Z","submitted_at":"2026-07-30T16:07:32Z","title":"AgentRadio: Passive Awareness for Long-Horizon Multi-Agent Collaboration","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-07-31T07:38:10.220810Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2607.28430"},"observation_digest":"sha256:c489f7b57f4876eb4e6037ea9b092699523b7bb88d0027623fbbacd5c8135aa1","observation_id":"f449c170-048b-4f46-b548-ab29ec9b13d4","resolution":{"observed_at":"2026-07-31T07:38:10.220810Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-03T00:48:02.213824Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.28661","last_updated":"2026-07-22T06:53:07Z","snapshot_observed_at":"2026-08-08T22:40:25.663112Z","submitted_at":"2026-07-22T06:53:07Z","title":"Are the Financial Reasoning from LLMs Credible? A Real World Test over Long-Horizon Statements","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-03T00:48:02.213824Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2607.28661"},"observation_digest":"sha256:d53cba98fa68e996320f15c16255de605e1ff817fccc856c3421c6a5254e7e6b","observation_id":"0b3933cf-56e0-441f-b9ee-af1d8ca75f4c","resolution":{"observed_at":"2026-08-03T00:48:02.213824Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-04T01:12:38.779498Z","title":"Finance agent benchmark: Benchmarking llms on real-world financial research tasks.arXiv preprint arXiv:2508.00828, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T23:09:41.455316Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:38.779498Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:7ff23ca049e643e6a7e73272a64caf210b67a894bf6d5e61be6a2622a9d34a8d","observation_id":"9f015896-0a35-463c-a30a-23394db146bc","resolution":{"observed_at":"2026-08-04T01:12:38.779498Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-05T00:22:58.807578Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00764","last_updated":"2026-08-01T16:54:41Z","snapshot_observed_at":"2026-08-09T13:32:46.703956Z","submitted_at":"2026-08-01T16:54:41Z","title":"FinDeepIndicator: Benchmarking Deep Research Agents in End-to-End Financial Indicator Construction","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-05T00:22:58.807578Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2608.00764"},"observation_digest":"sha256:7a6bcfe400f90d54508de2450bfc390a3676aba1b97ed070c3bae5b5ccaf2d27","observation_id":"19039dca-eb3f-4fa2-ae44-178c439954cd","resolution":{"observed_at":"2026-08-05T00:22:58.807578Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-07T22:11:21.458007Z","title":"Chen, H.; Narasimhan, K.; and Liu, Z","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.05864","last_updated":"2026-08-06T10:43:33Z","snapshot_observed_at":"2026-08-09T23:11:44.937820Z","submitted_at":"2026-08-06T10:43:33Z","title":"Seeing Is Not Deciding: Can Multimodal LLMs Act as Effective CEOs?","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T22:11:21.458007Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2608.05864"},"observation_digest":"sha256:b69fe95667f24abf15721e2b6d4a232c18dc8da69ad4978ee82ef9569864c022","observation_id":"8562865e-7d16-481e-9088-a1b8ef50899c","resolution":{"observed_at":"2026-08-07T22:11:21.458007Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-07T14:10:34.896207Z","title":"Cai,Y.;Hao,Y.;Zhou,J.;etal.2025.BuildingSelf-Evolving AgentsviaExperience-DrivenLifelongLearning:AFrame- work and Benchmark.arXiv preprint arXiv:2508.19005","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.06144","last_updated":"2026-08-06T15:14:54Z","snapshot_observed_at":"2026-08-10T03:43:34.971151Z","submitted_at":"2026-08-06T15:14:54Z","title":"FinEvo-Bench: A Longitudinal Benchmark for Self-Evolving Agents in Professional Financial Workflows","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-07T14:10:34.896207Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2608.06144"},"observation_digest":"sha256:b002d82fc305bfa06a8c63688784c1ed0b86097617f29158da6f0213782bbac0","observation_id":"dedac14f-73ee-4d02-9215-20738780a71a","resolution":{"observed_at":"2026-08-07T14:10:34.896207Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2508.00828/citation-record","integrity":"/paper/2508.00828/integrity","json":"/paper/2508.00828/citation-record.json","paper":"/paper/2508.00828"},"outbound":[],"paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","latest_version":1,"primary_category":"cs.CE","snapshot_observed_at":"2026-08-08T20:38:19.024001Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 32 inbound Pith citation observations for arXiv:2508.00828."}