{"as_of":"2026-08-18T14:27:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c35247f5e7737af44a76c2a13aa4a26557facbc6adc613c432cb165cf75ad0ae","coverage":[{"denominator":97,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":97,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T16:09:59.965344Z","state":"measured"},{"denominator":102,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":102,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":5,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":5,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T05:55:44.614190Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"cited_work":{"arxiv_id":"2509.09734","doi":"10.48550/arxiv.2509.09734","metadata_source":"arxiv_reference","pith_arxiv_id":"2509.09734","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mcp-agentbench: Evaluating real-world language agent performance with mcp-mediated tools","venue":"ArXiv.org","work_id":"03f7fc10-4b16-4fa7-8e0e-3a08b4133bc2","year":2025},"citing_paper":{"arxiv_id":"2602.10139","last_updated":"2026-04-26T01:34:23Z","snapshot_observed_at":"2026-08-14T10:04:04.434798Z","submitted_at":"2026-02-08T15:50:04Z","title":"Anonymization-Enhanced Privacy Protection for Mobile GUI Agents: Available but Invisible","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-16T05:59:32.583411Z"},"links":{"cited_paper":"/paper/2509.09734","citing_paper":"/paper/2602.10139"},"observation_digest":"sha256:3c7f33804e1826f4d0d128a30161bf30c6e08ccf8cd50a282ef205335ded88ff","observation_id":"46516960-3898-4c92-b35a-c2d1002de80c","resolution":{"observed_at":"2026-05-16T06:00:40.667290Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.09734","snapshot_observed_at":"2026-08-04T05:55:44.614190Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.05637","last_updated":"2026-08-03T15:19:02Z","snapshot_observed_at":"2026-08-07T17:01:33.563160Z","submitted_at":"2026-03-05T19:47:26Z","title":"Real Faults in Model Context Protocol (MCP) Software: a Comprehensive Taxonomy","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-04T05:55:44.614190Z"},"links":{"cited_paper":"/paper/2509.09734","citing_paper":"/paper/2603.05637"},"observation_digest":"sha256:b2eddc8ef09e29f44d8783f99cb7f6f164b969d0a5dd34e75681544275b87b1e","observation_id":"56e8193a-3f1c-40d5-9304-72274f6d06f6","resolution":{"observed_at":"2026-08-04T05:55:44.614190Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"cited_work":{"arxiv_id":"2509.09734","doi":"10.48550/arxiv.2509.09734","metadata_source":"arxiv_reference","pith_arxiv_id":"2509.09734","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mcp-agentbench: Evaluating real-world language agent performance with mcp-mediated tools","venue":"ArXiv.org","work_id":"03f7fc10-4b16-4fa7-8e0e-3a08b4133bc2","year":2025},"citing_paper":{"arxiv_id":"2604.03976","last_updated":"2026-05-04T20:58:24Z","snapshot_observed_at":"2026-07-06T22:53:01.835843Z","submitted_at":"2026-04-05T05:42:20Z","title":"Quantifying Trust: Financial Risk Management for Trustworthy AI Agents","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-13T17:16:17.464937Z"},"links":{"cited_paper":"/paper/2509.09734","citing_paper":"/paper/2604.03976"},"observation_digest":"sha256:3998a2b9c91616cdc078b19bef18bad21152792248ec1b279b1c7010c5e72369","observation_id":"76a02012-7f93-47c7-853e-a62f37ecebef","resolution":{"observed_at":"2026-05-13T17:16:34.273960Z","resolver_source":"orphan_title_repair","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"cited_work":{"arxiv_id":"2509.09734","doi":"10.48550/arxiv.2509.09734","metadata_source":"arxiv_reference","pith_arxiv_id":"2509.09734","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mcp-agentbench: Evaluating real-world language agent performance with mcp-mediated tools","venue":"ArXiv.org","work_id":"03f7fc10-4b16-4fa7-8e0e-3a08b4133bc2","year":2025},"citing_paper":{"arxiv_id":"2604.07551","last_updated":"2026-04-08T19:53:26Z","snapshot_observed_at":"2026-08-15T04:36:06.031902Z","submitted_at":"2026-04-08T19:53:26Z","title":"MCP-DPT: A Defense-Placement Taxonomy and Coverage Analysis for Model Context Protocol Security","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-10T17:10:50.283791Z"},"links":{"cited_paper":"/paper/2509.09734","citing_paper":"/paper/2604.07551"},"observation_digest":"sha256:e8bbdff25a80e5cabed5bd06b32a6c101bacce4bd17a53a1f65ded6cfc122034","observation_id":"8143449b-8b9b-4f30-a53f-ec024858b5a8","resolution":{"observed_at":"2026-05-10T21:10:46.718657Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"cited_work":{"arxiv_id":"2509.09734","doi":"10.48550/arxiv.2509.09734","metadata_source":"arxiv_reference","pith_arxiv_id":"2509.09734","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mcp-agentbench: Evaluating real-world language agent performance with mcp-mediated tools","venue":"ArXiv.org","work_id":"03f7fc10-4b16-4fa7-8e0e-3a08b4133bc2","year":2025},"citing_paper":{"arxiv_id":"2605.14312","last_updated":"2026-05-14T03:23:51Z","snapshot_observed_at":"2026-08-02T14:41:28.516568Z","submitted_at":"2026-05-14T03:23:51Z","title":"Making OpenAPI Documentation Agent-Ready: Detecting Documentation and REST Smells with a Multi-Agent LLM System","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-15T02:44:01.980796Z"},"links":{"cited_paper":"/paper/2509.09734","citing_paper":"/paper/2605.14312"},"observation_digest":"sha256:a101aba3c2a17ac4b384d9c73d057dcf2b23ecafa61090e6b49b48e7c972c30e","observation_id":"f53f3936-eac1-4b42-9000-d6c08880459e","resolution":{"observed_at":"2026-05-15T02:48:33.931507Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2509.09734/citation-record","integrity":"/paper/2509.09734/integrity","json":"/paper/2509.09734/citation-record.json","paper":"/paper/2509.09734"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-17T09:58:46.058102Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-15T16:09:59.575503Z","title":"Gpt-4 technical report.arXiv preprint arXiv:2303.08774, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.575503Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:cec926c079acd2a6cc14c6ff0228c44d0f4cc2453903021e634ba5df337ac31b","observation_id":"f04dbfa6-bd77-4fa5-a37a-4a79c0534a81","resolution":{"observed_at":"2026-08-15T16:09:59.575503Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.579805Z","title":"Introducing computer use, a new claude 3.5 sonnet, and claude 3.5 haiku","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.579805Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:27f790ec238349aa51b388600ef9a81f456636d71af494cca54a6022a91ab85d","observation_id":"1c176bf0-9594-424c-85a3-f87a194761ba","resolution":{"observed_at":"2026-08-15T16:09:59.579805Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.583800Z","title":"Claude 3.7 sonnet anthropic.https://www.anthropic.com/claude/sonnet, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.583800Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:06cb26b3c16b2c817ad7a47af074b4a64b7fb27d43a4cb5dcacee711fbb7b049","observation_id":"d86f00c7-2f53-4b4f-98f1-81525ec2864f","resolution":{"observed_at":"2026-08-15T16:09:59.583800Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.588628Z","title":"Introducing the model context protocol anthropic","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.588628Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:f90b714e325bb7a4300713a3b5b5dadeb9ec71d0fc5070d61dd778d87ea6d91b","observation_id":"13fd310f-916e-492d-8eb7-c9e0726846d5","resolution":{"observed_at":"2026-08-15T16:09:59.588628Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.592747Z","title":"chatmcp/mcprouter: api router for mcp servers.https://github.com/chatmcp/mcprouter, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.592747Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:fae8b5609c72531a816bb02371c41b03490afa5ed54529485de0f54a6ed1ca4a","observation_id":"168e0b27-54d9-4bf3-9214-81fc55a441e2","resolution":{"observed_at":"2026-08-15T16:09:59.592747Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06261","last_updated":"2025-12-19T14:25:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-07T17:36:04Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.06261","snapshot_observed_at":"2026-08-15T16:09:59.597405Z","title":"Gemini 2.5: Pushing the frontier with advanced reasoning, multimodality, long context, and next generation agentic capabilities.arXiv preprint arXiv:2507.06261, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.597405Z"},"links":{"cited_paper":"/paper/2507.06261","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:e30a0d9b41d098428f66d848df5fd636caaa0987d5f93d73cbf1bbfffd433cff","observation_id":"879c22a6-6d85-44f4-94b4-f24bbb89cca9","resolution":{"observed_at":"2026-08-15T16:09:59.597405Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.601414Z","title":"Deepseek-v3 technical report, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.601414Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:0e0b4cb4670200cbae3e2b6b667ec56b800322c301b51d01a9dcce61e577796f","observation_id":"0d707125-de91-4dd9-ab94-c8e7357ff80c","resolution":{"observed_at":"2026-08-15T16:09:59.601414Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.605512Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.605512Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:9cfb1b85ed8f5f2cfdef7056212dcfc97f506c7334e8ee751915111b4fb0d2f4","observation_id":"a69e02be-1b26-40b3-a1a4-625b84f0f6d9","resolution":{"observed_at":"2026-08-15T16:09:59.605512Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.608363Z","title":"Mcp-radar: A multi-dimensional benchmark for evaluating tool use capabilities in large language models.arXiv preprint arXiv:2505.16700, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.608363Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:3beae6f108836d063edb2eb2cbb00878c7c9377389585532a81c607f4cddb4ce","observation_id":"4868df1c-a4c8-4185-9104-61bdfeb4cd00","resolution":{"observed_at":"2026-08-15T16:09:59.608363Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01680","last_updated":"2024-04-19T01:15:16Z","snapshot_observed_at":"2026-08-10T13:10:07.804621Z","submitted_at":"2024-01-21T23:36:14Z","title":"Large Language Model based Multi-Agents: A Survey of Progress and Challenges","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01680","snapshot_observed_at":"2026-08-15T16:09:59.612031Z","title":"Large language model based multi-agents: A survey of progress and challenges.arXiv preprint arXiv:2402.01680, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.612031Z"},"links":{"cited_paper":"/paper/2402.01680","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:923bd9086f237606d3f19cf66ee369d4a57b2fce3e8d283fa6aaa6f5c2d7388b","observation_id":"47ac5a24-933a-4641-a266-90082df9a8ae","resolution":{"observed_at":"2026-08-15T16:09:59.612031Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.615822Z","title":"MetaGPT: Meta programming for a multi-agent collaborative framework","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.615822Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:57e6ce891c4f8369896d89af21f704f4eec766c47df7feae3cd12389ced76550","observation_id":"b7f07f99-ff07-4187-b9aa-fb238c678cfe","resolution":{"observed_at":"2026-08-15T16:09:59.615822Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23278","last_updated":"2025-10-07T07:13:32Z","snapshot_observed_at":"2026-08-14T03:34:08.418318Z","submitted_at":"2025-03-30T01:58:22Z","title":"Model Context Protocol (MCP): Landscape, Security Threats, and Future Research Directions","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.23278","snapshot_observed_at":"2026-08-15T16:09:59.619700Z","title":"Model context protocol (mcp): Landscape, security threats, and future research directions.arXiv preprint arXiv:2503.23278, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.619700Z"},"links":{"cited_paper":"/paper/2503.23278","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:5a603226ea858d396eddc0e0e91316b340cd92cc97567071b3726d5a9eac891b","observation_id":"b6663164-3117-4521-a806-277fa4e005d3","resolution":{"observed_at":"2026-08-15T16:09:59.619700Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.10403","last_updated":"2023-05-26T17:59:33Z","snapshot_observed_at":"2026-08-06T19:23:30.079167Z","submitted_at":"2022-12-20T16:29:03Z","title":"Towards Reasoning in Large Language Models: A Survey","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.10403","snapshot_observed_at":"2026-08-15T16:09:59.623636Z","title":"Towards reasoning in large language models: A survey.arXiv preprint arXiv:2212.10403, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.623636Z"},"links":{"cited_paper":"/paper/2212.10403","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:5d1382aee78782c167bfe3a96e8e4b94fb464ce74cda5941781dc29598a08db5","observation_id":"06a0485c-5b00-4639-8864-5c4b1bd3625a","resolution":{"observed_at":"2026-08-15T16:09:59.623636Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.02716","last_updated":"2024-02-05T04:25:24Z","snapshot_observed_at":"2026-08-15T04:53:46.192738Z","submitted_at":"2024-02-05T04:25:24Z","title":"Understanding the planning of LLM agents: A survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.02716","snapshot_observed_at":"2026-08-15T16:09:59.627932Z","title":"Understanding the planning of llm agents: A survey.arXiv preprint arXiv:2402.02716, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.627932Z"},"links":{"cited_paper":"/paper/2402.02716","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:cc53d619a09ff1e578012b896b08c4f269091782fc089a847f7007e7ff90d888","observation_id":"07521cd2-d296-4137-989a-a8f50c0fc1f5","resolution":{"observed_at":"2026-08-15T16:09:59.627932Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.631917Z","title":"Api-bank: A comprehensive benchmark for tool-augmented llms, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.631917Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:8c6d0041fd732975b05dafc1cd6a86ba715d407fb442644aec34185f76f6f705","observation_id":"6bcab3ce-d232-43fd-9c9f-8e146d990f11","resolution":{"observed_at":"2026-08-15T16:09:59.631917Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.02977","last_updated":"2025-12-03T03:33:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-04T15:59:41Z","title":"Large Language Model-Based Agents for Software Engineering: A Survey","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.02977","snapshot_observed_at":"2026-08-15T16:09:59.635780Z","title":"Large language model-based agents for software engineering: A survey.arXiv preprint arXiv:2409.02977, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.635780Z"},"links":{"cited_paper":"/paper/2409.02977","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:e0e1c6988a92dc9990acab5a887b57ea9ab7781701ed12e3b86bff03fb7f3fb4","observation_id":"d062aab7-dc66-429e-9918-2665bfa91a1c","resolution":{"observed_at":"2026-08-15T16:09:59.635780Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.12806","last_updated":"2025-08-01T22:37:16Z","snapshot_observed_at":"2026-08-09T01:25:07.357969Z","submitted_at":"2025-07-17T05:46:27Z","title":"MCPEval: Automatic MCP-based Deep Evaluation for AI Agent Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.12806","snapshot_observed_at":"2026-08-15T16:09:59.639299Z","title":"Mcpeval: Automatic mcp-based deep evaluation for ai agent models.arXiv preprint arXiv:2507.12806, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.639299Z"},"links":{"cited_paper":"/paper/2507.12806","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:e2812420998ff4d7d1387ca64e0682a35829bac3affc56745141520ac9c3209d","observation_id":"b57c8d05-524a-4962-90ca-703cb0946f1a","resolution":{"observed_at":"2026-08-15T16:09:59.639299Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.11094","last_updated":"2025-04-18T10:39:23Z","snapshot_observed_at":"2026-08-16T12:40:56.136470Z","submitted_at":"2025-04-15T11:40:12Z","title":"Evaluation Report on MCP Servers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.11094","snapshot_observed_at":"2026-08-15T16:09:59.643148Z","title":"Evaluation report on mcp servers.arXiv preprint arXiv:2504.11094, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.643148Z"},"links":{"cited_paper":"/paper/2504.11094","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:b1bb6419c5b4c48ef34cfca40a4f6a22fcd9c9a2b7211059cd5be40e344f1903","observation_id":"0e22d4aa-a2ec-4a60-b40b-c55e0199e61d","resolution":{"observed_at":"2026-08-15T16:09:59.643148Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07945","last_updated":"2024-02-09T02:33:45Z","snapshot_observed_at":"2026-08-16T14:20:01.793685Z","submitted_at":"2024-02-09T02:33:45Z","title":"ScreenAgent: A Vision Language Model-driven Computer Control Agent","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07945","snapshot_observed_at":"2026-08-15T16:09:59.648057Z","title":"Screenagent: A vision language model-driven computer control agent.arXiv preprint arXiv:2402.07945, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.648057Z"},"links":{"cited_paper":"/paper/2402.07945","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:2aa93dbd37e153440521a0b58fae52f921b810a072380dd1462473f2f6a4023a","observation_id":"0e4953a0-bd75-46eb-bedd-8f58f5056c7e","resolution":{"observed_at":"2026-08-15T16:09:59.648057Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.652027Z","title":"Hello gpt-4o | openai.https://openai.com/index/hello-gpt-4o/, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.652027Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:48feb8d4206edb6a2a55246765f1107f08f99cafe96e514f9b1e6e5829336373","observation_id":"f1533033-293f-4d9e-b5e3-81f3a4df788f","resolution":{"observed_at":"2026-08-15T16:09:59.652027Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.655974Z","title":"Openai o3-mini | openai.https://openai.com/index/openai-o3-mini/, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.655974Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:7bbbacbab094c97ba825e41a651de931e312bb1f1f77dffbaf7f98ace8613c72","observation_id":"7a2132bd-0960-4387-99d3-732c6b30046a","resolution":{"observed_at":"2026-08-15T16:09:59.655974Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.659667Z","title":"Llm rankings | openrouter.https://openrouter.ai/rankings?view=month, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.659667Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:1872d2122fcf028a7a9b661a73f29784c20090c3bad061c3060e3186233b9e53","observation_id":"205ea629-1552-464c-be80-0b4bf5ab1ad1","resolution":{"observed_at":"2026-08-15T16:09:59.659667Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12373","last_updated":"2024-07-16T06:19:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-18T07:58:33Z","title":"WebCanvas: Benchmarking Web Agents in Online Environments","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12373","snapshot_observed_at":"2026-08-15T16:09:59.663790Z","title":"Webcanvas: Benchmarking web agents in online environments.arXiv preprint arXiv:2406.12373, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.663790Z"},"links":{"cited_paper":"/paper/2406.12373","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:a3ad4d9496fa336e0c2e4b0c67f09935f5ee3c599494cd3f9926be44e1aeb3ba","observation_id":"b23ac9c7-101b-4720-9d94-62a6f025607f","resolution":{"observed_at":"2026-08-15T16:09:59.663790Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.668957Z","title":"Gorilla: Large language model connected with massive apis.Advances in Neural Information Processing Systems, 37:126544–126565, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.668957Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:2a048117986daf2ca3955a7fa6049f98bcf3704dd442ed5e62a7fe7800a79fca","observation_id":"cf9b569d-ad23-46f3-ba78-3c0be6bb8eee","resolution":{"observed_at":"2026-08-15T16:09:59.668957Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.671862Z","title":"Toolllm: Facilitating large language models to master 16000+ real-world apis, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.671862Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:21c60679b3cc8e042e32e2f361fce52333655bc93bcf26106db8dc86d9e5e5ba","observation_id":"b6f77a3c-da88-4678-9a86-5ab38b87d8c1","resolution":{"observed_at":"2026-08-15T16:09:59.671862Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.674685Z","title":"Language agents: Foundations, prospects, and risks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.674685Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:d1a03131ab48ab3f22df519e4d0e8811ee4158fa0b7a3cb1b918c1c429a5d3d2","observation_id":"8fb6bc00-b3df-4228-93fc-aed0701492c5","resolution":{"observed_at":"2026-08-15T16:09:59.674685Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.678632Z","title":"Cognitive architectures for language agents.Transactions on Machine Learning Research, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.678632Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:08010c8448c8f99870eccd9d6c73c4bb940ce5f3be5e537cfe142f4be3e1cfa9","observation_id":"07f89f4e-6fc2-4cdb-a061-2977d0d97121","resolution":{"observed_at":"2026-08-15T16:09:59.678632Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.20534","last_updated":"2026-02-03T04:57:00Z","snapshot_observed_at":"2026-08-16T14:37:33.548231Z","submitted_at":"2025-07-28T05:35:43Z","title":"Kimi K2: Open Agentic Intelligence","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.20534","snapshot_observed_at":"2026-08-15T16:09:59.682902Z","title":"Kimi k2: Open agentic intelligence.arXiv preprint arXiv:2507.20534, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.682902Z"},"links":{"cited_paper":"/paper/2507.20534","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:48c7b59a65589f704284f4e5d27c0d60ad8af7c4d25c4162659f92c30692b5e8","observation_id":"c28884ec-3641-494b-b103-79fdd299ada3","resolution":{"observed_at":"2026-08-15T16:09:59.682902Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.687387Z","title":"Qwen3 technical report, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.687387Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:5a8a1113c60bf61ac41148339152d14c5a71c1ca03c22de41859c18f4d7594d0","observation_id":"6fc6353a-0c45-4e43-9e29-6a556f706414","resolution":{"observed_at":"2026-08-15T16:09:59.687387Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.690716Z","title":"A survey on large language model based autonomous agents.Frontiers of Computer Science, 18(6):186345, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.690716Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:499211cecdfe3e90849d47962a2db8b7cbbfcc89d147388185e66acfe61c80e4","observation_id":"f96705e4-3d83-40c3-877e-48ac8aba597f","resolution":{"observed_at":"2026-08-15T16:09:59.690716Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.694872Z","title":"Autogen: Enabling next-gen llm applications via multi-agent conversation, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.694872Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:5b96d16b12b3f94f5fc8c486e7099d6c38b51806d9b41aeb7de0265c757ef859","observation_id":"14b090d7-c4c3-426b-8210-f1e21ef8ae1c","resolution":{"observed_at":"2026-08-15T16:09:59.694872Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.698092Z","title":"Osworld: Benchmarking multimodal agents for open-ended tasks in real computer environments.Advances in Neural Information Processing Systems, 37:52040–52094, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.698092Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:ea2f3aa29cb66da80dcf75495358d7ae8a2aca70208cfe41ba19ba6123a76fde","observation_id":"044d3413-1474-4962-abc7-413accfe3377","resolution":{"observed_at":"2026-08-15T16:09:59.698092Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.701522Z","title":"An illusion of progress? assessing the current state of web agents.arXiv preprint arXiv:2504.01382, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.701522Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:9eebc5c383a5c28090c4267ec6a0136fc1f55075d6d40e099d1af20588c86edb","observation_id":"30ad9d7a-68b4-4ef4-9ef4-1fcb6118cc0d","resolution":{"observed_at":"2026-08-15T16:09:59.701522Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.861579Z","title":"Patil, Ion Stoica, and Joseph E","venue":null,"work_id":"80c68fce-0a7d-4679-bd70-28dc412bd766","year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.705683Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:59d69def0f0a9fb9d4e042bf683a9b110fffe589393e50d1aa6f63a3bee6e370","observation_id":"47498be3-a365-4029-9a3e-7b47fdecd2a5","resolution":{"observed_at":"2026-08-15T16:10:00.865496Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.709630Z","title":"Swe-agent: Agent-computer interfaces enable automated software engineering.Advances in Neural Information Processing Systems, 37:50528–50652, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.709630Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:89e1b56608069632d46ed347c2e20e5ff0c4b51763acc6f2539c652f82229122","observation_id":"62bb34c7-8896-46fb-b289-a2fea6208e5c","resolution":{"observed_at":"2026-08-15T16:09:59.709630Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12045","last_updated":"2024-06-17T19:33:08Z","snapshot_observed_at":"2026-08-17T20:31:29.818313Z","submitted_at":"2024-06-17T19:33:08Z","title":"$\\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12045","snapshot_observed_at":"2026-08-15T16:09:59.714886Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.714886Z"},"links":{"cited_paper":"/paper/2406.12045","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:46c76e63c5588db7548b8314f38761d408be4c4aff606c73f5eed5f51983cd29","observation_id":"f8d13c9c-1275-4834-8797-61d36ac0433e","resolution":{"observed_at":"2026-08-15T16:09:59.714886Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.18279","last_updated":"2025-05-06T15:08:00Z","snapshot_observed_at":"2026-08-17T10:40:45.253009Z","submitted_at":"2024-11-27T12:13:39Z","title":"Large Language Model-Brained GUI Agents: A Survey","version":12},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.18279","snapshot_observed_at":"2026-08-15T16:09:59.718887Z","title":"Large language model-brained gui agents: A survey.arXiv preprint arXiv:2411.18279, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.718887Z"},"links":{"cited_paper":"/paper/2411.18279","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:2bdaec544dd6c7bb5064c68da2b82a34cfe10b8f1598279275ee0275ec9c8376","observation_id":"c0642870-3162-40f6-973f-62484b96e688","resolution":{"observed_at":"2026-08-15T16:09:59.718887Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.722052Z","title":"Judging llm-as-a-judge with mt-bench and chatbot arena.Advances in Neural Information Processing Systems, 36:46595–46623, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.722052Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:4fcee31cb66286bac52f7a0623a46bde9cdcdd5b507110ab9ae5ff5c0b006f3b","observation_id":"37da5206-ca5d-4fd5-a2cd-18bf17786bef","resolution":{"observed_at":"2026-08-15T16:09:59.722052Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.840798Z","title":"Complexfuncbench: Exploring multi-step and constrained function calling under long-context scenario, 2025","venue":null,"work_id":"7ba60250-b51b-454a-842c-166089967ff9","year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.725222Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:af22d648bd3c082a80b7963aa0d15cf8949fe553b0962b0e68eda1ea53f9ad69","observation_id":"d4982fd5-e71f-4645-9f2b-453ae62a5819","resolution":{"observed_at":"2026-08-15T16:10:00.844547Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.13854","last_updated":"2024-04-16T15:13:18Z","snapshot_observed_at":"2026-08-14T11:14:55.351653Z","submitted_at":"2023-07-25T22:59:32Z","title":"WebArena: A Realistic Web Environment for Building Autonomous Agents","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.13854","snapshot_observed_at":"2026-08-15T16:09:59.729245Z","title":"low-pass-rate","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.729245Z"},"links":{"cited_paper":"/paper/2307.13854","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:2368311948296c30705f5f2901d7ca31bbcfa390bbd4b0afe65a4fa7e0a496b4","observation_id":"063cfff0-3a05-4f1d-8618-a3f802bab087","resolution":{"observed_at":"2026-08-15T16:09:59.729245Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.832060Z","title":null,"venue":null,"work_id":"7fb5c920-564b-4e8a-b0b3-352aec192fad","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.732808Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:d3927a7244f6a149d1eadb1f172991c59d7b1d93954e93463ba11a46b20e9d66","observation_id":"5d63803d-fe75-4be9-b0bc-702cebfdbc14","resolution":{"observed_at":"2026-08-15T16:10:00.834958Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.811925Z","title":null,"venue":null,"work_id":"59ea35a6-e557-4dc3-89f0-20982b54b84c","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.739668Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:268c1cee9254705275b1789f41d1c472a74aa502925b2ae7932a3523f64aef9e","observation_id":"50ead5b2-c716-43c4-8504-8bd1f0e6f188","resolution":{"observed_at":"2026-08-15T16:10:00.815240Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.801778Z","title":"this weekend","venue":null,"work_id":"51bba25e-3706-40f9-b9a5-279648adfcbc","year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.743339Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:380f17cdfa0f2766a28293a939e062e5bc7c89e2b00934ddbd2786e68d29785f","observation_id":"f5c7a093-9bbb-4acf-8f77-4f14248ce584","resolution":{"observed_at":"2026-08-15T16:10:00.804961Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.791507Z","title":null,"venue":null,"work_id":"b4ed306b-f315-4148-a8ae-a61528f224cf","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.746948Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:0d48a16ff48725402acd9caccc42b77aea082324a8e27982c457ac98ebde487f","observation_id":"0872d06d-f242-441f-8d5a-7492a3e83ad5","resolution":{"observed_at":"2026-08-15T16:10:00.795014Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.781756Z","title":null,"venue":null,"work_id":"40b9e521-53b4-441b-bdae-88bb5951eb1d","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.750875Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:9557c9238ec94e43c45f93f2479b4b1f1c911d771d73d878e092e2acd3f3508d","observation_id":"5f60a7dc-3ab1-485f-9421-1160cac72b58","resolution":{"observed_at":"2026-08-15T16:10:00.785292Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.772230Z","title":null,"venue":null,"work_id":"2721b942-9d49-4e8c-a197-b05c54d8dcee","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.755547Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:03020eabcd2efac9a7689ae42e63e9a63801d24b7610719dce9e6d03e014b94a","observation_id":"417c3555-9d27-4221-a08c-a629b8caff02","resolution":{"observed_at":"2026-08-15T16:10:00.775369Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.761679Z","title":null,"venue":null,"work_id":"166154bf-47cb-4836-8437-5de65c52cae2","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.759952Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:4395c2094fe4bfac91bf970741a2b8357f4a1c5cfd068c024ab161800798e5d3","observation_id":"d1b3d1ca-c724-4e2c-8680-6c976401d345","resolution":{"observed_at":"2026-08-15T16:10:00.765669Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.751236Z","title":null,"venue":null,"work_id":"ff0b530e-71a8-45e6-b7fc-143d69a2f008","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.764247Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:d911e7a4b33dac75dc1d4a1fa6ec6aa3c8a8237d692d26f1b835c823d7ec4902","observation_id":"55b850f5-ea9a-430d-aa06-ca4b086e4429","resolution":{"observed_at":"2026-08-15T16:10:00.755150Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.740456Z","title":null,"venue":null,"work_id":"79a444d0-4327-4c0a-82c6-d8ddee7b6ca3","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.768616Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:25adad91452f2f06a4f136ac2e8eacf79d9f45b3a335ca9b92ab49fe1ac465a8","observation_id":"58542e5c-d3b3-4f89-98b9-788b22e529fe","resolution":{"observed_at":"2026-08-15T16:10:00.744955Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.731833Z","title":null,"venue":null,"work_id":"68283b6e-1f0d-4f36-a890-d6ffb033caa2","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.771695Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:f2a79ce4d651d39147bebeb42ee1acc8d677afc5f4c56537ec9edb22d183c5c7","observation_id":"4240e96b-e641-428f-b72a-0cb0fe4d9d21","resolution":{"observed_at":"2026-08-15T16:10:00.734739Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.717676Z","title":null,"venue":null,"work_id":"c95c642d-b363-4dd2-a3ef-d0a8decbce6b","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.775130Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:82c9802e79b78bb92ab1c6dcf5f0d97c78afda1c92c878d55725f172dfa5f099","observation_id":"89bf7a33-d3d8-40a8-b2bb-40449aab706a","resolution":{"observed_at":"2026-08-15T16:10:00.725461Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.707593Z","title":null,"venue":null,"work_id":"815f3a59-f21c-4896-81bd-2527e6c0accc","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.779420Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:e43e6838d9f12445bbb52cdc44717125b7b408b2b69db0f6b6021e9b0fc7ba08","observation_id":"b77ee098-db42-4074-a3f0-327b8376719e","resolution":{"observed_at":"2026-08-15T16:10:00.710547Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.696882Z","title":null,"venue":null,"work_id":"257f43dd-8cfc-4104-a7da-a4f38a5e292e","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.783120Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:41938e629e17dc4e6e70d899f2dbe1b25477aaa10e7c427ce08fd0ad74202c4c","observation_id":"ece40590-03c0-47a4-b3fa-432c743b8615","resolution":{"observed_at":"2026-08-15T16:10:00.700614Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.687440Z","title":null,"venue":null,"work_id":"2bdf91c2-e68b-41a2-831d-43a239dfeafd","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.787685Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:b128d50b93c849b4e059499084ac4b7159aca1cab0a1fe0b03b63dcc035c97ad","observation_id":"a6c449b7-9933-47c4-a29b-d0b96ba0c29b","resolution":{"observed_at":"2026-08-15T16:10:00.690565Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.674606Z","title":null,"venue":null,"work_id":"af7a6b01-2611-48de-af52-d6f77849f7be","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.792288Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:862272e84952fc7dbc856af9a5a39b77cbcc101efb7dcbf4a95651e6d758f393","observation_id":"5c7c3dc5-a122-4fb3-af84-35aec7076c99","resolution":{"observed_at":"2026-08-15T16:10:00.678315Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.665240Z","title":"Contextual Component Generation System Prompt You are a Realistic MCP Server Tool Scenario Designer","venue":null,"work_id":"f1b5cdab-5c89-4b7b-8556-764fd1a8f461","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.796437Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:2942b423777e511476704f705714027daafb0c5ba8eca3c89fd2a90232a5dfce","observation_id":"7e5d1bab-289c-4e38-b10b-0e3077876a7e","resolution":{"observed_at":"2026-08-15T16:10:00.668751Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.655068Z","title":"Explain why these tools are necessary and sufficient","venue":null,"work_id":"f455bf77-65f9-42c9-843f-4ae8187aef0f","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.799863Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:ec71f0b0e775c9f2ccc49e91333c39bedf498bb965088630225d2da28aff7664","observation_id":"c697fc60-be5d-42fd-a47d-018c3d214872","resolution":{"observed_at":"2026-08-15T16:10:00.658691Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.645997Z","title":"</user_profile>","venue":null,"work_id":"77760619-e97e-40c1-b1ef-98b51f8a8ba4","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.803218Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:5101367531a35461553c7bca38c60e121cbdcaf9c9217bd936a7fe944fdbee2c","observation_id":"abd66cc0-994c-4349-b839-d97ee7fe5178","resolution":{"observed_at":"2026-08-15T16:10:00.648996Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.636771Z","title":"</scenario>","venue":null,"work_id":"5558de6b-87d7-4d0b-9b4c-f7ec1ca6ada8","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.806551Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:30a1e3bf043de685e5f9ad91914bcdbca12fa494b135f449254a4e6d0ae80956","observation_id":"ed6153d1-d3f3-4303-b612-f4f45a947cca","resolution":{"observed_at":"2026-08-15T16:10:00.639882Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.626685Z","title":"</objective> Parameter Sourcing Requirements All tool parameters must come from:","venue":null,"work_id":"c900ca1a-df7a-4592-9550-2934a9bafe3f","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.809854Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:42600fc776b217f65c38d1273d6830d75676b54242e624044e09641781613ea3","observation_id":"985203c1-be12-4404-bb49-fe827d4242b4","resolution":{"observed_at":"2026-08-15T16:10:00.629853Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.616581Z","title":null,"venue":null,"work_id":"45ff5167-2b98-4c30-8c7d-544d18cd0d68","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.814043Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:f9d0e052780910f0a4a1e1450ed7800205beed2e5acd570499a7d4afecb07c38","observation_id":"da2982c8-d296-49f4-83c4-816f00e39d0b","resolution":{"observed_at":"2026-08-15T16:10:00.619600Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.823114Z","title":null,"venue":null,"work_id":"189248b7-c927-4d48-b388-c942b765c7f2","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.818146Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:dda6021f37842db32889771b0c5eb678ad8720762c5e8c218785754c7a001830","observation_id":"b707530b-fdb7-4c48-aea1-bbde2172ee01","resolution":{"observed_at":"2026-08-15T16:10:00.826281Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.605043Z","title":"All necessary information must be available in the initial request or derived from tool usage","venue":null,"work_id":"a6d0448c-b192-4a9e-a9f6-260a6f3f487c","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.821334Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:758f44a8f954e863c4e98623fea2bd15919f369e9e0847b64ee4d88d98f302bc","observation_id":"6516e0c3-8ae9-4065-b601-88e630b6ceaa","resolution":{"observed_at":"2026-08-15T16:10:00.608870Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.593307Z","title":null,"venue":null,"work_id":"f3c011b1-e3ef-41ff-acea-acc6912bc05b","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.824797Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:e9f7f83c72609edbdf98c8981f401016d5eeedd2ccb968bca2fcbc4593997ef4","observation_id":"2557775e-36cc-4a04-97d8-784974b4c29b","resolution":{"observed_at":"2026-08-15T16:10:00.597445Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.583231Z","title":null,"venue":null,"work_id":"bd9d536d-c75b-4767-b337-fda8109e774a","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.828077Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:5ddd03a95483bec8ccfed983cbab65c2ce5a942d63f9331de7ef7eb6f62080e9","observation_id":"1bb2157d-dfaf-464c-90c0-50b8e6f87b0d","resolution":{"observed_at":"2026-08-15T16:10:00.586246Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.573368Z","title":null,"venue":null,"work_id":"c25d077d-52b3-4513-99cd-84f4bb9da995","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.830910Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:58c97e431dee61e9b58ce17caf2ada5c8114f0479b8481a65139fa458b10beb5","observation_id":"657a37d3-6f32-4958-8e1a-3ee772e0cf2f","resolution":{"observed_at":"2026-08-15T16:10:00.577060Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.563083Z","title":"18 ReAcT Assistant Prompt You are an advanced AI assistant with access to Model Context Protocol (MCP) servers","venue":null,"work_id":"f36fb59d-6fac-4d95-b2cc-93827dd9843f","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.834090Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:4de6949b707bff74f6fe57fece8dfa2d07ae7c92a0c8f2fb961aec8e02456516","observation_id":"83c73c7c-e788-4558-a8d5-ef4a9ad8a69f","resolution":{"observed_at":"2026-08-15T16:10:00.566663Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.529267Z","title":null,"venue":null,"work_id":"0f8c66e5-66d4-4322-9bf0-7b6bd06df0ea","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.844340Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:c9a1bc27ac7a43e3a5f9886b0deaf4dd3b2074cd3d81f9fca91517bad70bb915","observation_id":"229d5c6e-0714-427d-949d-7cb44f0453a1","resolution":{"observed_at":"2026-08-15T16:10:00.532847Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.453336Z","title":null,"venue":null,"work_id":"ff1af438-ad47-4eec-bb3a-1b9585e83f26","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.870421Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:a5e8647ffd1105c785c05cfa82552a3e42d3837d5b4dddceedf0937c921abcf5","observation_id":"1d356728-faec-46e2-8d6a-2ee761e6afa7","resolution":{"observed_at":"2026-08-15T16:10:00.457398Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.431126Z","title":"Use tools strategically but don’t overcomplicate simple requests that can be answered directly","venue":null,"work_id":"32a02202-ee2e-4844-bf3e-3c129fe82f97","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.876280Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:a3d7d6b5628e2568ac8ba8943ab6c037a89fa2b03e8ecccaf1ab5a7a1b1f5078","observation_id":"5cd8c327-a81c-4e8d-b2be-4d1b440e7ff7","resolution":{"observed_at":"2026-08-15T16:10:00.435982Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.421742Z","title":null,"venue":null,"work_id":"c78d6fba-40e6-4d91-8368-182278d38251","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.878985Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:8e7b0c3de0d4f474a09d78bb2c3c39db8afad2d4da308cb998d8bc954f79694f","observation_id":"4043924b-b56e-4208-b72c-077fad4a032e","resolution":{"observed_at":"2026-08-15T16:10:00.425198Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.411120Z","title":null,"venue":null,"work_id":"9912e2e6-fb3b-4a12-89f9-bc7f25704cdc","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.881951Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:923ee1b23da7cee5c81da528660eac49c02dcae1d81fe66771a0533734a728e0","observation_id":"e1829e18-510d-4851-bdaa-c7c0d8b168f0","resolution":{"observed_at":"2026-08-15T16:10:00.414636Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.400977Z","title":null,"venue":null,"work_id":"74f134f4-5a77-4a06-b267-bf596ef10433","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.884910Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:9a53979cd1ca891d4de3c6442991bb3b002c3589cdefbf5e5e9398d843c222bd","observation_id":"a83a8bf4-61af-4858-b53f-3c8fee1578d6","resolution":{"observed_at":"2026-08-15T16:10:00.404351Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.391589Z","title":"calculate","venue":null,"work_id":"1b09f79f-bbd8-47c5-950d-3186119b8496","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.888074Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:46ababc131878e542c68e55c4f9b7a8b84a42295db4a511b850f66bd01b0ad53","observation_id":"a1663766-b26c-43c2-bbc6-329752d2741d","resolution":{"observed_at":"2026-08-15T16:10:00.395335Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.382066Z","title":null,"venue":null,"work_id":"7f1b88ae-ec75-4bfe-bc2c-ec050acbf913","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.891084Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:8af6c98d04d4e6955d13750a1fb39eaf18114d91f53cfee1473852047d4c4736","observation_id":"a89b75c7-6712-4809-867a-fe0623828ba6","resolution":{"observed_at":"2026-08-15T16:10:00.385190Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.372095Z","title":"name\": \"selected_tool_name","venue":null,"work_id":"7f3c1b8f-46c2-4d37-8257-571279dee0e0","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.894774Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:c13fed2332f0883581092f8287bee686010e88d4735a44073c23e2367420ee32","observation_id":"70dc9e5c-e84f-4938-8385-896eb8c5b574","resolution":{"observed_at":"2026-08-15T16:10:00.375172Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.550302Z","title":null,"venue":null,"work_id":"a131ef5a-07c5-4110-ad92-fc2a649af4ee","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.898845Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:4c127fb72287bec4b4da0422f0b7ba19675cc070153e63326b9461e44ef9a344","observation_id":"6b993c42-c56e-4c77-a97a-ccdac8625a57","resolution":{"observed_at":"2026-08-15T16:10:00.554806Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.540178Z","title":null,"venue":null,"work_id":"3aeaa6a7-3126-4918-b55e-476dd6477af8","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.901865Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:50808d95ec1e4c8d7e50689d08a8e5c55f7e3619bd802e3b1173c15401e09f93","observation_id":"97ab0760-1241-4af9-9c9f-26a9a314982d","resolution":{"observed_at":"2026-08-15T16:10:00.544081Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.362524Z","title":null,"venue":null,"work_id":"c55969a7-58c5-4f79-b8c5-cb0a43024e24","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.905292Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:6d8e7d70c6908b35dffd92b4bb5e0c7703fdc0a988e6574c3dd6bbe8ffe336e1","observation_id":"e0fb5e69-d851-460a-aa9a-b6ecb96aa77f","resolution":{"observed_at":"2026-08-15T16:10:00.365791Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.519103Z","title":null,"venue":null,"work_id":"c9b04d0e-0677-4be1-b49d-a78d4a556306","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.908836Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:4984a530cab675b97c6289ba5b88f0e92d43215b91b972b6c2520b0d7f4e88db","observation_id":"c30e4b07-e76a-45ff-9b89-249153480e2a","resolution":{"observed_at":"2026-08-15T16:10:00.522858Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.506985Z","title":null,"venue":null,"work_id":"f6b5a472-8c12-439d-b5d5-0194231d4f33","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.911505Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:683960336d993e698cbcf8ec7604a495b26c9e365022d46a303a840059bb06de","observation_id":"cf0e8efc-d0b2-4d7f-a7a2-c0daed721ae5","resolution":{"observed_at":"2026-08-15T16:10:00.510747Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.496309Z","title":null,"venue":null,"work_id":"25e4467d-03b7-46d1-aaaa-0a907277eab4","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.914609Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:e33adb7ab21682e6d598e638b5a11b6db94fd6f39b77a1b712ac3e6a01b66f81","observation_id":"10e81301-20ef-4921-8e66-420614433afc","resolution":{"observed_at":"2026-08-15T16:10:00.499240Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.487091Z","title":null,"venue":null,"work_id":"3671732d-89f1-45a4-a2d1-98b6243a5cb6","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.917497Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:a2dec605a3d47f67e47cd3ace282163cedca3bf1831972dc38036d734ec34d7a","observation_id":"ebd9e01b-b608-48c8-8200-8ca908dad167","resolution":{"observed_at":"2026-08-15T16:10:00.490476Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.477963Z","title":null,"venue":null,"work_id":"f053a554-54c9-43ef-9e11-4992a4c36a78","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.920809Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:314c45d5d2bd2037b1f231b7ba0f8fc9fe9a326b1ad7b0dcb2024c035c3d925b","observation_id":"da615eed-1126-4ee7-bc81-ed4c88104657","resolution":{"observed_at":"2026-08-15T16:10:00.480993Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.466197Z","title":null,"venue":null,"work_id":"dcca3ce8-b978-498a-a522-26506b9d5d31","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.923510Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:75bb37a8beab41fb10aa969775bb63b7fa4f63f0005698d311d26099b1627d3f","observation_id":"7acaa1f0-89ab-458c-be71-7ed9df5c4008","resolution":{"observed_at":"2026-08-15T16:10:00.470719Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.352505Z","title":null,"venue":null,"work_id":"b55700a3-f385-4706-b568-72758c67e256","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.926469Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:d32e52ff3f888a60b5c43e4aca2182de61446d7626838b252d5344f8c7f415a6","observation_id":"6c08c922-68d2-4611-8db4-4fa37716a2b6","resolution":{"observed_at":"2026-08-15T16:10:00.355565Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.443237Z","title":null,"venue":null,"work_id":"81e29e6c-ec49-4128-9bb4-03f7bb961b04","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.929765Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:d6d9bf2059d675d70527a68b31dadb0eba4bac272f4da5eb606180f04f911a5a","observation_id":"83bc4bf8-18b4-42ea-ba74-8b8cfc6e65fb","resolution":{"observed_at":"2026-08-15T16:10:00.447203Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.340784Z","title":"Use tools strategically but don’t overcomplicate simple requests that can be answered directly","venue":null,"work_id":"f9534cc9-00ca-4b8e-8a6a-5f79a49709fb","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.933367Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:94810e339d744f5c729cec2785cb2934899c97d196b1c03f55934cad9a285a41","observation_id":"22ba905f-b340-4c0d-9088-a5b4825eb36b","resolution":{"observed_at":"2026-08-15T16:10:00.344957Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.330088Z","title":"typically,","venue":null,"work_id":"1f7f6a28-71e9-423e-82f7-fb08581653fb","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.937178Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:34ba12a7db80a4fb09bd294d847bd4256e4a466b8e062c7fff7baf186af14a60","observation_id":"2b55d9f9-b424-4265-a1ab-a285d7e48451","resolution":{"observed_at":"2026-08-15T16:10:00.333928Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.319479Z","title":"proof\" or","venue":null,"work_id":"81c2a349-6744-4fbb-b96a-43eb154fe7a8","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.940243Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:0e5d6f6476381079a59081466ab10d28bd344ee8ef421d66c7f2e8b1ec68fab8","observation_id":"29e93816-391c-42a4-8d64-4db7d28a3d90","resolution":{"observed_at":"2026-08-15T16:10:00.323241Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.308452Z","title":"Verification details","venue":null,"work_id":"6b240919-3dae-499a-bb60-566c7e198587","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":101,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.943340Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:dc732ee741d47946fdc3d2da256c0294f5e54f39705c47d67e2fd8ee21d1392d","observation_id":"fea0b363-9c63-4e45-877e-6b34cb413f67","resolution":{"observed_at":"2026-08-15T16:10:00.311853Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.296963Z","title":null,"venue":null,"work_id":"9058b07b-9952-4d35-a976-d7a7e53e99ed","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":102,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.947012Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:de48d7b4e26b8c82f4e657ed0d12bc3c240948c6b6c3dc04c54e5c88bda500f3","observation_id":"5d597700-554b-4119-abdf-bab294d4c826","resolution":{"observed_at":"2026-08-15T16:10:00.300334Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.286219Z","title":null,"venue":null,"work_id":"d166641c-4d79-410d-b5d0-ec525cdf1026","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":103,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.950392Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:1018611cb4e87379b7178b726a8cadb7b3aee9ff64fdcdcf4c1c8f0cd498a55a","observation_id":"87ac6600-c55e-4873-ae41-d263e5e79900","resolution":{"observed_at":"2026-08-15T16:10:00.289757Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.275469Z","title":"Lacks verification details about data source","venue":null,"work_id":"0ecb60f4-81bf-48d3-b042-f8baecc649e3","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":104,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.954473Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:7e8c29d4737b6f72ede903d8f05caf00f8afe94c6b3da30056c9403bea952b8b","observation_id":"096281c7-054f-47e6-94ae-0470684d2cd1","resolution":{"observed_at":"2026-08-15T16:10:00.278836Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.264723Z","title":"• Process explanations • Formatting differences 5.Decide: • External data + Core need met = PASS • Knowledge only = FAIL","venue":null,"work_id":"8b70ba01-5e2b-471a-8a86-53ee40ab66dc","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":105,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.957941Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:00d085abbdb42e48568a2cac02cc6273c35a47613644c7eb1574095590949f3c","observation_id":"77a03be7-8536-4173-93a7-86b25ce8b96e","resolution":{"observed_at":"2026-08-15T16:10:00.269044Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.251031Z","title":"Is there specific external data that helps the user?","venue":null,"work_id":"83d57746-86dc-46d7-b1d3-abd8978f56bb","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":106,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.961419Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:bbab3bf2be728ae7f66756f6695b0de524a4e11ec734b307726ab73c65dffb32","observation_id":"2e1e93ba-941f-4c01-b977-d593ba76da6e","resolution":{"observed_at":"2026-08-15T16:10:00.256273Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.239776Z","title":"nice-to-have","venue":null,"work_id":"f91c5079-dc4b-4e43-a323-123c018c6620","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":107,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.965344Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:12072580109405ebd6d738a7f8f48c8c53d38f8011f132210d3b0308a502a366","observation_id":"6f305a3b-9d9f-43e6-a629-a1a386f1f675","resolution":{"observed_at":"2026-08-15T16:10:00.243841Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools"},"reference_resolution":{"displayed":97,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":76,"verified_exact":0,"verified_fuzzy":21},"total_outbound_references":97},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 97 of 97 outbound references and 5 inbound Pith citation observations for arXiv:2509.09734."}