{"as_of":"2026-08-09T15:51:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b2f6280a50be673945a0cb3d766d947dcce669b497cd9c5570bb64e316c20b65","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":43,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":43,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":43,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":43,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T20:11:11.989677Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T21:18:59.715072Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":"2310.03128","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-07-03T21:18:59.715072Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use","venue":null,"work_id":"b02ef874-7bf1-4042-b075-bed194bafa8e","year":2023},"citing_paper":{"arxiv_id":"2401.05561","last_updated":"2024-09-30T10:17:12Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-01-10T22:07:21Z","title":"TrustLLM: Trustworthiness in Large Language Models","version":6},"reference_index":140,"source":"pdf_text","source_observed_at":"2026-05-18T11:17:08.108565Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2401.05561"},"observation_digest":"sha256:22f2dc85319207185bcbef5c9e89df5b9e2b2b2d5232a14d4a1760d1d7cbdd27","observation_id":"bfdcb510-4117-4c31-bdb4-3897e0a75dd1","resolution":{"observed_at":"2026-05-18T11:17:08.521093Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":"2310.03128","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-07-03T21:18:59.715072Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use","venue":null,"work_id":"b02ef874-7bf1-4042-b075-bed194bafa8e","year":2023},"citing_paper":{"arxiv_id":"2402.17177","last_updated":"2024-04-17T18:41:39Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-27T03:30:58Z","title":"Sora: A Review on Background, Technology, Limitations, and Opportunities of Large Vision Models","version":3},"reference_index":122,"source":"pdf_text","source_observed_at":"2026-05-13T13:43:11.024069Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2402.17177"},"observation_digest":"sha256:20c88aeb60dc04cbd02eb54e1064f9bfab6172da3ab3b628ea1bc9aea8214d43","observation_id":"15ec5b6d-4d56-46f8-a061-989c9f6bc2bc","resolution":{"observed_at":"2026-05-13T13:43:11.181459Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":"2310.03128","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-07-03T21:18:59.715072Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use","venue":null,"work_id":"b02ef874-7bf1-4042-b075-bed194bafa8e","year":2023},"citing_paper":{"arxiv_id":"2406.12045","last_updated":"2024-06-17T19:33:08Z","snapshot_observed_at":"2026-08-08T21:08:39.676079Z","submitted_at":"2024-06-17T19:33:08Z","title":"$\\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-11T03:19:00.831153Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2406.12045"},"observation_digest":"sha256:cace947a34b8d31e430a74ee0a870fd6ccc9cd6ba9ce128bd31fb6cfd44ea780","observation_id":"55495220-d2ad-4deb-8d59-2736b19284fe","resolution":{"observed_at":"2026-05-11T03:19:00.864649Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-08-07T20:11:11.989677Z","title":"Z., et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.09897","last_updated":"2025-02-14T04:07:25Z","snapshot_observed_at":"2026-08-09T12:17:42.951426Z","submitted_at":"2025-02-14T04:07:25Z","title":"Artificial Intelligence in Spectroscopy: Advancing Chemistry from Prediction to Generation and Beyond","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T20:11:11.989677Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2502.09897"},"observation_digest":"sha256:1b5871f738930fe4b85c20dcebcf9b195360617b82b85005b8319d1aeddec098","observation_id":"34424258-f287-4c29-95a8-71d578960cf5","resolution":{"observed_at":"2026-08-07T20:11:11.989677Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":"2310.03128","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-07-03T21:18:59.715072Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use","venue":null,"work_id":"b02ef874-7bf1-4042-b075-bed194bafa8e","year":2023},"citing_paper":{"arxiv_id":"2504.19793","last_updated":"2025-08-24T03:28:21Z","snapshot_observed_at":"2026-08-01T07:24:09.967062Z","submitted_at":"2025-04-28T13:36:43Z","title":"Prompt Injection Attack to Tool Selection in LLM Agents","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-16T17:08:28.933831Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2504.19793"},"observation_digest":"sha256:fb28caf98a352334d82f1b0add545087a9bf29c32e35a0835fc52769873d5fb9","observation_id":"c18c7bb9-1a8c-4176-9085-053352f1ff88","resolution":{"observed_at":"2026-05-16T17:08:28.977781Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":"2310.03128","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-07-03T21:18:59.715072Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use","venue":null,"work_id":"b02ef874-7bf1-4042-b075-bed194bafa8e","year":2023},"citing_paper":{"arxiv_id":"2506.07982","last_updated":"2025-06-09T17:52:18Z","snapshot_observed_at":"2026-07-06T21:39:13.304260Z","submitted_at":"2025-06-09T17:52:18Z","title":"$\\tau^2$-Bench: Evaluating Conversational Agents in a Dual-Control Environment","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-12T07:52:17.174347Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2506.07982"},"observation_digest":"sha256:2ba84df9bece8a428842508688ea0e2d4abf04578dd3d22954e01004e8422224","observation_id":"9e16af64-e263-45dc-b0d4-27ecf863a911","resolution":{"observed_at":"2026-05-12T07:52:17.470715Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-08-07T04:58:45.870083Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use.arXiv preprint arXiv:2310.03128,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09173","last_updated":"2025-06-10T18:38:57Z","snapshot_observed_at":"2026-08-07T04:53:04.454979Z","submitted_at":"2025-06-10T18:38:57Z","title":"The Curious Language Model: Strategic Test-Time Information Acquisition","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T04:58:45.870083Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2506.09173"},"observation_digest":"sha256:0687f08b619d77501e2e40f31e7ee3520e228645ecc3e281ec5275b6052c63dc","observation_id":"b59b1997-5a74-45d7-8a59-a4d8437cf4cc","resolution":{"observed_at":"2026-08-07T04:58:45.870083Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-08-06T22:02:38.861022Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.22853","last_updated":"2025-07-02T07:55:09Z","snapshot_observed_at":"2026-08-07T20:47:09.014588Z","submitted_at":"2025-06-28T11:28:04Z","title":"DICE-BENCH: Evaluating the Tool-Use Capabilities of Large Language Models in Multi-Round, Multi-Party Dialogues","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-06T22:02:38.861022Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2506.22853"},"observation_digest":"sha256:2770018ac3f02c8bedbb39c7cbed7d55d346ac9b0810675d3e988c80e036fadc","observation_id":"20756517-6596-4bec-9837-d139ff5bdd0b","resolution":{"observed_at":"2026-08-06T22:02:38.861022Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-08-06T21:48:47.549550Z","title":"arXiv preprint arXiv:2310.03128","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.23394","last_updated":"2025-06-29T20:47:27Z","snapshot_observed_at":"2026-08-07T20:47:43.953021Z","submitted_at":"2025-06-29T20:47:27Z","title":"Teaching a Language Model to Speak the Language of Tools","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T21:48:47.549550Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2506.23394"},"observation_digest":"sha256:5cdab61cfd4133913b56b4f4c3555d450778eeee8a8309775567d827d239761e","observation_id":"6ac1df20-9361-4c43-82c9-f4898d7e2d0e","resolution":{"observed_at":"2026-08-06T21:48:47.549550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-08-06T21:22:43.754386Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use // arXiv preprint arXiv:2310.03128","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.00487","last_updated":"2025-07-02T04:35:44Z","snapshot_observed_at":"2026-08-07T22:30:31.357864Z","submitted_at":"2025-07-01T07:02:26Z","title":"MassTool: A Multi-Task Search-Based Tool Retrieval Framework for Large Language Models","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T21:22:43.754386Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2507.00487"},"observation_digest":"sha256:ab48aa6e6360601f95d3ca80370921ff6cab2b8b6d0a8a2c5b6bb3e4b6ed36fc","observation_id":"337c765f-ae53-452a-8937-39caf61300d2","resolution":{"observed_at":"2026-08-06T21:22:43.754386Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-08-06T15:39:57.024887Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15296","last_updated":"2025-07-21T06:55:37Z","snapshot_observed_at":"2026-08-09T08:28:26.487927Z","submitted_at":"2025-07-21T06:55:37Z","title":"Butterfly Effects in Toolchains: A Comprehensive Analysis of Failed Parameter Filling in LLM Tool-Agent Systems","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T15:39:57.024887Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2507.15296"},"observation_digest":"sha256:16d46d92491f12b6a7ffc4e647b7860eb5d29213fe227e0ea501167a0bb1c13f","observation_id":"0b1ca4a9-c1c7-4390-8c31-1e095790d3e2","resolution":{"observed_at":"2026-08-06T15:39:57.024887Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-08-06T15:20:38.656905Z","title":"Yue Huang, Jiawen Shi, Yuan Li, Chenrui Fan, Siyuan Wu, Qihui Zhang, Yixin Liu, Pan Zhou, Yao Wan, Neil Zhenqiang Gong, and Lichao Sun","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.16217","last_updated":"2025-08-29T18:45:22Z","snapshot_observed_at":"2026-08-08T13:31:49.745762Z","submitted_at":"2025-07-22T04:21:03Z","title":"Towards Compute-Optimal Many-Shot In-Context Learning","version":2},"reference_index":2001,"source":"pdf_text","source_observed_at":"2026-08-06T15:20:38.656905Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2507.16217"},"observation_digest":"sha256:8253e7f22bafce02e5317131f0ac193b2e9d78f3419f6ca822a42ddf80fed532","observation_id":"22550fcd-f160-4d2f-bbd9-269ad5bc33d1","resolution":{"observed_at":"2026-08-06T15:20:38.656905Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":"2310.03128","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-07-03T21:18:59.715072Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use","venue":null,"work_id":"b02ef874-7bf1-4042-b075-bed194bafa8e","year":2023},"citing_paper":{"arxiv_id":"2507.21035","last_updated":"2026-05-17T21:43:43Z","snapshot_observed_at":"2026-07-06T22:04:01.721229Z","submitted_at":"2025-07-28T17:55:08Z","title":"GenoMAS: A Multi-Agent Framework for Scientific Discovery via Code-Driven Gene Expression Analysis","version":3},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-22T00:37:11.945418Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2507.21035"},"observation_digest":"sha256:fada3f7a436c58a964cd85f21984144081773df517949100033b6b1db8690e09","observation_id":"51babeb4-227a-4516-95b0-2f094cd24194","resolution":{"observed_at":"2026-05-22T00:40:51.463887Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-08-06T12:44:21.609937Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.21504","last_updated":"2025-07-29T04:57:02Z","snapshot_observed_at":"2026-08-06T15:35:42.279155Z","submitted_at":"2025-07-29T04:57:02Z","title":"Evaluation and Benchmarking of LLM Agents: A Survey","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T12:44:21.609937Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2507.21504"},"observation_digest":"sha256:9d9ae8d170b184a71cf07e9170a53327981385d7331bf89c635097f5325f7d35","observation_id":"1576480d-e593-4139-8f1a-353f60d92f52","resolution":{"observed_at":"2026-08-06T12:44:21.609937Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-08-06T12:09:26.335428Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.22034","last_updated":"2025-07-29T17:34:12Z","snapshot_observed_at":"2026-08-07T20:29:47.386254Z","submitted_at":"2025-07-29T17:34:12Z","title":"UserBench: An Interactive Gym Environment for User-Centric Agents","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-06T12:09:26.335428Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2507.22034"},"observation_digest":"sha256:ff3c4a7d6ba91b360560ee708c3d73bd4ead937e2209b2102d8f754afadc5256","observation_id":"67a0b2f9-4580-4f30-9e34-68085d9089bd","resolution":{"observed_at":"2026-08-06T12:09:26.335428Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-08-05T18:46:16.475327Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.14323","last_updated":"2025-08-25T04:46:11Z","snapshot_observed_at":"2026-08-08T11:00:39.506759Z","submitted_at":"2025-08-20T00:35:50Z","title":"Beyond Semantic Similarity: Reducing Unnecessary API Calls via Behavior-Aligned Retriever","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-05T18:46:16.475327Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2508.14323"},"observation_digest":"sha256:58721d61dc77e5c8d93b61b453a48b4d0fac73f69826a273dc70785db862be9a","observation_id":"915671c3-e5cc-4ec2-9679-5d17a1a45d50","resolution":{"observed_at":"2026-08-05T18:46:16.475327Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":"2310.03128","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-07-03T21:18:59.715072Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use","venue":null,"work_id":"b02ef874-7bf1-4042-b075-bed194bafa8e","year":2023},"citing_paper":{"arxiv_id":"2510.02837","last_updated":"2026-05-25T02:56:44Z","snapshot_observed_at":"2026-08-09T14:58:53.653794Z","submitted_at":"2025-10-03T09:19:15Z","title":"Beyond the Final Answer: Evaluating the Reasoning Trajectories of Tool-Augmented Agents","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-18T11:02:55.529271Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2510.02837"},"observation_digest":"sha256:a443e37342fc8941cb355e6caed8ac06552786cb4027d1bdf0acd23d3ec74dd0","observation_id":"a03ba34e-3ea1-4c44-a50e-849e2df3cda7","resolution":{"observed_at":"2026-05-18T11:06:17.792208Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-08-04T12:44:12.592641Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use.arXiv preprint arXiv:2310.03128,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.02837","last_updated":"2026-05-25T02:56:44Z","snapshot_observed_at":"2026-08-09T14:58:53.653794Z","submitted_at":"2025-10-03T09:19:15Z","title":"Beyond the Final Answer: Evaluating the Reasoning Trajectories of Tool-Augmented Agents","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-04T12:44:12.592641Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2510.02837"},"observation_digest":"sha256:e7a67127fab4974f6e86cd1016b30540087ca27ec7fa194a7d181a2dd048de08","observation_id":"0e9d9af4-cf15-48a8-824e-b99be0ce07c2","resolution":{"observed_at":"2026-08-04T12:44:12.592641Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":"2310.03128","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-07-03T21:18:59.715072Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use","venue":null,"work_id":"b02ef874-7bf1-4042-b075-bed194bafa8e","year":2023},"citing_paper":{"arxiv_id":"2510.23853","last_updated":"2026-04-15T18:39:35Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T20:51:58Z","title":"Your LLM Agents are Temporally Blind: The Misalignment Between Tool Use Decisions and Human Time Perception","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-18T03:46:03.228969Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2510.23853"},"observation_digest":"sha256:58aed5eb95d18888fc212c3ee2ed761240c4b2b2f79f4e5f81e26bf3f6cafdfd","observation_id":"523d99e1-82e0-4b96-84b8-3501b8c110a9","resolution":{"observed_at":"2026-05-18T03:50:52.539483Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":"2310.03128","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-07-03T21:18:59.715072Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use","venue":null,"work_id":"b02ef874-7bf1-4042-b075-bed194bafa8e","year":2023},"citing_paper":{"arxiv_id":"2512.13564","last_updated":"2026-01-13T09:33:57Z","snapshot_observed_at":"2026-08-06T08:27:07.254588Z","submitted_at":"2025-12-15T17:22:34Z","title":"Memory in the Age of AI Agents","version":2},"reference_index":261,"source":"arxiv_source","source_observed_at":"2026-05-11T18:18:19.911342Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2512.13564"},"observation_digest":"sha256:c8af8ead00f0ec42e93caf64b22dc7f56bc7308b0505becdbb9544fdf3bf4c9e","observation_id":"0700dca1-1a6d-46f8-8cd4-7a0dabd4525f","resolution":{"observed_at":"2026-05-11T18:18:20.250588Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-08-03T09:21:36.280577Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2601.14192","last_updated":"2026-07-03T11:26:58Z","snapshot_observed_at":"2026-08-07T15:05:40.772054Z","submitted_at":"2026-01-20T17:51:56Z","title":"Toward Efficient Agents: Memory, Tool learning, and Planning","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-03T09:21:36.280577Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2601.14192"},"observation_digest":"sha256:f65645b9bb9555c784e1f1064c91539878b539734ae4e96d909d22cfcf8e71b4","observation_id":"ad97d443-4774-437f-b967-c37f48d6ce05","resolution":{"observed_at":"2026-08-03T09:21:36.280577Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-08-02T21:46:33.268448Z","title":"Z., and Sun, L","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.19218","last_updated":"2026-06-09T15:02:00Z","snapshot_observed_at":"2026-08-05T17:29:43.215311Z","submitted_at":"2026-02-22T15:02:00Z","title":"Gecko: A Simulation Environment with Stateful Feedback for Refining Agent Tool Calls","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-02T21:46:33.268448Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2602.19218"},"observation_digest":"sha256:f9eb093ae0f97ef704edea15d8defe652840a5c2ecf0d3a02414a287fdaff8e7","observation_id":"59ec2ede-71a8-4033-86a9-01bfcebf3c9f","resolution":{"observed_at":"2026-08-02T21:46:33.268448Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-07-14T19:59:59.030585Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2604.00137","last_updated":"2026-07-10T22:46:45Z","snapshot_observed_at":"2026-08-04T07:28:22.927886Z","submitted_at":"2026-03-31T18:42:36Z","title":"Open, Reliable, and Collective: A Community-Driven Framework for Tool-Using AI Agents","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-07-14T19:59:59.030585Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2604.00137"},"observation_digest":"sha256:8198bfe5ef4afd1f2b18bcbc7958c534a34c67312bde0121d931b3b7554f188d","observation_id":"4b853e52-1d2d-404e-83a7-6d6775680c58","resolution":{"observed_at":"2026-07-14T19:59:59.030585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":"2310.03128","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-07-03T21:18:59.715072Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use","venue":null,"work_id":"b02ef874-7bf1-4042-b075-bed194bafa8e","year":2023},"citing_paper":{"arxiv_id":"2605.00737","last_updated":"2026-08-06T13:58:03Z","snapshot_observed_at":"2026-08-09T14:11:44.621025Z","submitted_at":"2026-05-01T15:38:13Z","title":"To Call or Not to Call: A Framework to Assess and Optimize LLM Tool Calling","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-05-09T19:32:57.054584Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2605.00737"},"observation_digest":"sha256:b31c2a847c846232a5d0482acc4475137bb3cbd380bbc9c07cdb97da168a4ebb","observation_id":"4f2bdf27-338b-487a-8c00-d32a047c8fd4","resolution":{"observed_at":"2026-05-11T15:36:10.333342Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":"2310.03128","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-07-03T21:18:59.715072Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use","venue":null,"work_id":"b02ef874-7bf1-4042-b075-bed194bafa8e","year":2023},"citing_paper":{"arxiv_id":"2605.02411","last_updated":"2026-07-31T18:57:52Z","snapshot_observed_at":"2026-08-06T23:11:08.908082Z","submitted_at":"2026-05-04T10:01:24Z","title":"FitText: Evolving Agent Tool Ecologies via Memetic Retrieval","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-05-08T19:04:53.639217Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2605.02411"},"observation_digest":"sha256:709695b1c0d9d9172f2cfa2da2b51aa63f1b2c36e23e1ccc14d5d9445539428c","observation_id":"4a5476e6-df85-4c89-9155-3d38d16c87f8","resolution":{"observed_at":"2026-05-09T06:05:33.837524Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":"2310.03128","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-07-03T21:18:59.715072Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use","venue":null,"work_id":"b02ef874-7bf1-4042-b075-bed194bafa8e","year":2023},"citing_paper":{"arxiv_id":"2605.02411","last_updated":"2026-07-31T18:57:52Z","snapshot_observed_at":"2026-08-06T23:11:08.908082Z","submitted_at":"2026-05-04T10:01:24Z","title":"FitText: Evolving Agent Tool Ecologies via Memetic Retrieval","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-07-01T00:25:46.507401Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2605.02411"},"observation_digest":"sha256:08284adff7a7ba97636ec394e8b5d99cb8db5e9f53054165527f10953566613a","observation_id":"d00cbb03-51f7-491b-9c00-d4a541116888","resolution":{"observed_at":"2026-07-01T00:45:12.424114Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-08-04T05:24:44.710671Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2605.02411","last_updated":"2026-07-31T18:57:52Z","snapshot_observed_at":"2026-08-06T23:11:08.908082Z","submitted_at":"2026-05-04T10:01:24Z","title":"FitText: Evolving Agent Tool Ecologies via Memetic Retrieval","version":3},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-04T05:24:44.710671Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2605.02411"},"observation_digest":"sha256:95c19ec65592eaf5575a45f2070c7ca4a4bd5cc844b2bdcf01c35e41d4dd08d8","observation_id":"f87b296a-6676-48f1-9d59-9f30edbfe7a2","resolution":{"observed_at":"2026-08-04T05:24:44.710671Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":"2310.03128","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-07-03T21:18:59.715072Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use","venue":null,"work_id":"b02ef874-7bf1-4042-b075-bed194bafa8e","year":2023},"citing_paper":{"arxiv_id":"2605.02682","last_updated":"2026-05-04T15:00:37Z","snapshot_observed_at":"2026-08-03T00:40:33.048580Z","submitted_at":"2026-05-04T15:00:37Z","title":"Hybrid Inspection and Task-Based Access Control in Zero-Trust Agentic AI","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-08T19:06:36.952203Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2605.02682"},"observation_digest":"sha256:87c7a37604d29505c36356fca333f0c7eb3345c75808691ad11dc4581d6ed44e","observation_id":"7709fb7c-2ad0-4ef3-886e-0853e6383a01","resolution":{"observed_at":"2026-05-09T06:00:37.672983Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":"2310.03128","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-07-03T21:18:59.715072Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use","venue":null,"work_id":"b02ef874-7bf1-4042-b075-bed194bafa8e","year":2023},"citing_paper":{"arxiv_id":"2605.03986","last_updated":"2026-05-05T17:08:26Z","snapshot_observed_at":"2026-07-06T23:16:49.553583Z","submitted_at":"2026-05-05T17:08:26Z","title":"From Intent to Execution: Composing Agentic Workflows with Agent Recommendation","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-07T16:19:39.392516Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2605.03986"},"observation_digest":"sha256:6aeeb3dd77f6f041e69a6ac25ac8df43ebd9853a74258ae9849d91d10de02ed6","observation_id":"684b6e09-e7cb-4db7-9017-2bb70f976afd","resolution":{"observed_at":"2026-05-11T23:46:42.967662Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":"2310.03128","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-07-03T21:18:59.715072Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use","venue":null,"work_id":"b02ef874-7bf1-4042-b075-bed194bafa8e","year":2023},"citing_paper":{"arxiv_id":"2605.07251","last_updated":"2026-05-08T05:19:59Z","snapshot_observed_at":"2026-07-06T23:19:35.885430Z","submitted_at":"2026-05-08T05:19:59Z","title":"Can Agents Price a Reaction? Evaluating LLMs on Chemical Cost Reasoning","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-11T01:29:47.384341Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2605.07251"},"observation_digest":"sha256:8a1de87b7dc7e90586a89bb57bde3b914dc47144125c4de074f0b25bdb74cb41","observation_id":"a86b0777-73cc-4fea-906b-4dfb1a7f260f","resolution":{"observed_at":"2026-05-11T01:45:51.785672Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":"2310.03128","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-07-03T21:18:59.715072Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use","venue":null,"work_id":"b02ef874-7bf1-4042-b075-bed194bafa8e","year":2023},"citing_paper":{"arxiv_id":"2605.14038","last_updated":"2026-05-17T15:23:37Z","snapshot_observed_at":"2026-08-03T15:23:23.551096Z","submitted_at":"2026-05-13T18:59:28Z","title":"Model-Adaptive Tool Necessity Reveals the Knowing-Doing Gap in LLM Tool Use","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-15T05:31:54.252816Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2605.14038"},"observation_digest":"sha256:043af048d595dda643bc97d7d9d7be8aef5d3479ca34f355f9d6660032f79a5a","observation_id":"b053461d-5763-4d1d-8f70-7b549a050ee9","resolution":{"observed_at":"2026-05-15T05:35:04.476298Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":"2310.03128","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-07-03T21:18:59.715072Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use","venue":null,"work_id":"b02ef874-7bf1-4042-b075-bed194bafa8e","year":2023},"citing_paper":{"arxiv_id":"2605.14038","last_updated":"2026-05-17T15:23:37Z","snapshot_observed_at":"2026-08-03T15:23:23.551096Z","submitted_at":"2026-05-13T18:59:28Z","title":"Model-Adaptive Tool Necessity Reveals the Knowing-Doing Gap in LLM Tool Use","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-20T20:52:20.975459Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2605.14038"},"observation_digest":"sha256:02ab4eb1a9e1101a9ea58d105b3282258a1b80f378bb3e77e8339a21d884af07","observation_id":"3e545d30-a3b9-44e7-b915-75cc50001943","resolution":{"observed_at":"2026-05-20T20:53:43.552641Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":"2310.03128","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-07-03T21:18:59.715072Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use","venue":null,"work_id":"b02ef874-7bf1-4042-b075-bed194bafa8e","year":2023},"citing_paper":{"arxiv_id":"2605.14892","last_updated":"2026-05-15T09:12:41Z","snapshot_observed_at":"2026-08-07T13:04:48.115170Z","submitted_at":"2026-05-14T14:36:13Z","title":"Beyond Individual Intelligence: Surveying Collaboration, Failure Attribution, and Self-Evolution in LLM-based Multi-Agent Systems","version":1},"reference_index":164,"source":"arxiv_source","source_observed_at":"2026-05-15T03:07:38.232966Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2605.14892"},"observation_digest":"sha256:cc0a193e40087b08a302cb22b2e6cc98b681a398068d7c9671e3775ade0b6554","observation_id":"d473db4c-3e82-4cd3-9767-bcb8063c5b44","resolution":{"observed_at":"2026-05-15T03:08:57.998107Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":"2310.03128","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-07-03T21:18:59.715072Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use","venue":null,"work_id":"b02ef874-7bf1-4042-b075-bed194bafa8e","year":2023},"citing_paper":{"arxiv_id":"2605.14892","last_updated":"2026-05-15T09:12:41Z","snapshot_observed_at":"2026-08-07T13:04:48.115170Z","submitted_at":"2026-05-14T14:36:13Z","title":"Beyond Individual Intelligence: Surveying Collaboration, Failure Attribution, and Self-Evolution in LLM-based Multi-Agent Systems","version":2},"reference_index":165,"source":"arxiv_source","source_observed_at":"2026-05-19T16:51:13.491389Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2605.14892"},"observation_digest":"sha256:f470e2644c08d123dd8bc8c7fb5455af241cbeeb20c77c45fb963e6dd9bec241","observation_id":"749d96c9-e863-41ee-9a88-6548c8b301dd","resolution":{"observed_at":"2026-05-19T16:52:39.943407Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":"2310.03128","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-07-03T21:18:59.715072Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use","venue":null,"work_id":"b02ef874-7bf1-4042-b075-bed194bafa8e","year":2023},"citing_paper":{"arxiv_id":"2605.22177","last_updated":"2026-05-21T08:47:49Z","snapshot_observed_at":"2026-08-02T13:10:19.838403Z","submitted_at":"2026-05-21T08:47:49Z","title":"Maestro: Reinforcement Learning to Orchestrate Hierarchical Model-Skill Ensembles","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-22T07:51:13.362986Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2605.22177"},"observation_digest":"sha256:e404a2372f6e5816891409a1cb35d2bebfee3315e3d18cd25a2adccf23356e85","observation_id":"b36e908b-d4c6-4335-a441-73f48c1b2f6c","resolution":{"observed_at":"2026-05-22T07:51:15.401757Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":"2310.03128","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-07-03T21:18:59.715072Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use","venue":null,"work_id":"b02ef874-7bf1-4042-b075-bed194bafa8e","year":2023},"citing_paper":{"arxiv_id":"2605.26154","last_updated":"2026-05-24T04:26:13Z","snapshot_observed_at":"2026-07-06T23:36:01.564747Z","submitted_at":"2026-05-24T04:26:13Z","title":"MemMorph: Tool Hijacking in LLM Agents via Memory Poisoning","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-30T00:25:59.532059Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2605.26154"},"observation_digest":"sha256:a4dd7f47c87e10c3b2597405cbe1aa2f7e7b5419f5af1b6be50c8d28c132638b","observation_id":"b7301589-2913-4ce2-9e68-39c437d74e07","resolution":{"observed_at":"2026-07-01T16:35:50.732455Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":"2310.03128","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-07-03T21:18:59.715072Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use","venue":null,"work_id":"b02ef874-7bf1-4042-b075-bed194bafa8e","year":2023},"citing_paper":{"arxiv_id":"2606.00251","last_updated":"2026-05-29T18:32:14Z","snapshot_observed_at":"2026-08-02T17:00:44.481600Z","submitted_at":"2026-05-29T18:32:14Z","title":"Capability Self-Assessment: Teaching LLMs to Know Their Limits","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-28T22:25:50.579196Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2606.00251"},"observation_digest":"sha256:289a9118b39aabf41de2cbffd794cd28d2f51cf0f32a276893bb1d5c3d044c14","observation_id":"df94b234-2959-4db7-a73e-2469787000e9","resolution":{"observed_at":"2026-06-28T22:32:44.643921Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":"2310.03128","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-07-03T21:18:59.715072Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use","venue":null,"work_id":"b02ef874-7bf1-4042-b075-bed194bafa8e","year":2023},"citing_paper":{"arxiv_id":"2606.06566","last_updated":"2026-06-04T17:27:39Z","snapshot_observed_at":"2026-08-06T00:01:29.832904Z","submitted_at":"2026-06-04T17:27:39Z","title":"NTILC: Neural Tool Invocation via Learned Compression","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-28T00:05:36.600389Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2606.06566"},"observation_digest":"sha256:c717baa007b0b07c8a9f0bdb5b86f655dd42f55b940f188bf0d73372946aef4c","observation_id":"0cbf862b-fdcf-46a9-adff-cdc6cc2f28a2","resolution":{"observed_at":"2026-07-02T15:07:04.773446Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":"2310.03128","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-07-03T21:18:59.715072Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use","venue":null,"work_id":"b02ef874-7bf1-4042-b075-bed194bafa8e","year":2023},"citing_paper":{"arxiv_id":"2606.06667","last_updated":"2026-07-06T17:01:35Z","snapshot_observed_at":"2026-08-02T12:07:09.543257Z","submitted_at":"2026-06-04T19:32:00Z","title":"The Piggyback Hypothesis of Generalization: Explaining and Mitigating Emergent Misalignment","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-28T01:34:23.382705Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2606.06667"},"observation_digest":"sha256:1cd115b3efe3447b6cbd56102c41a8d316a3e5f5a64451992c38b87bc68fdcfb","observation_id":"7f6a9ba2-b68d-4f41-8fe9-02eb8f016c6e","resolution":{"observed_at":"2026-07-02T13:06:59.155289Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-07-12T14:59:04.152474Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use.arXiv preprint arXiv:2310.03128,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.06667","last_updated":"2026-07-06T17:01:35Z","snapshot_observed_at":"2026-08-02T12:07:09.543257Z","submitted_at":"2026-06-04T19:32:00Z","title":"The Piggyback Hypothesis of Generalization: Explaining and Mitigating Emergent Misalignment","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-12T14:59:04.152474Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2606.06667"},"observation_digest":"sha256:bb6bcf69888e93c93821763035abf455b059d868a60a82a0afdd3e2487e3bbbf","observation_id":"c152ec48-6f8b-4634-82ed-6532c4cc7091","resolution":{"observed_at":"2026-07-12T14:59:04.152474Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":"2310.03128","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-07-03T21:18:59.715072Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use","venue":null,"work_id":"b02ef874-7bf1-4042-b075-bed194bafa8e","year":2023},"citing_paper":{"arxiv_id":"2606.18051","last_updated":"2026-06-16T15:27:55Z","snapshot_observed_at":"2026-08-07T03:31:46.971110Z","submitted_at":"2026-06-16T15:27:55Z","title":"Compositional Skill Routing for LLM Agents: Decompose, Retrieve, and Compose","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-06-27T00:37:50.570757Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2606.18051"},"observation_digest":"sha256:5f34a0c9e08cab43867be8f3f1d1d6743f028f90a6db795fa7fea30a9dff3d56","observation_id":"b490de2a-c3a7-4ef9-8b45-7932e70decb2","resolution":{"observed_at":"2026-07-03T21:18:59.716659Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-08-01T01:35:09.657698Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.25765","last_updated":"2026-07-28T14:19:59Z","snapshot_observed_at":"2026-08-09T00:29:07.819944Z","submitted_at":"2026-07-28T14:19:59Z","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-01T01:35:09.657698Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2607.25765"},"observation_digest":"sha256:35fb5841ae7eb1ce90b2519b6e6cb22a2ec9fe556ab925850a275b5aaeec07a9","observation_id":"cf78261e-76a9-4938-8b66-59b2e68a263f","resolution":{"observed_at":"2026-08-01T01:35:09.657698Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-08-03T00:55:29.777251Z","title":"arXiv preprint arXiv:2310.03128 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.28636","last_updated":"2026-05-19T13:56:13Z","snapshot_observed_at":"2026-08-06T00:38:04.006327Z","submitted_at":"2026-05-19T13:56:13Z","title":"Chain-of-Models: Cross-Model Auditing for Bias-Robust LLM Judges","version":1},"reference_index":156,"source":"arxiv_source","source_observed_at":"2026-08-03T00:55:29.777251Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2607.28636"},"observation_digest":"sha256:c6368efcbcd92e973bcfb0694d13a2fb0c5e65d7811c74c9bb27418dc21bad96","observation_id":"f5bd2c1c-a64a-4a8f-bc81-35180af2a342","resolution":{"observed_at":"2026-08-03T00:55:29.777251Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2310.03128/citation-record","integrity":"/paper/2310.03128/integrity","json":"/paper/2310.03128/citation-record.json","paper":"/paper/2310.03128"},"outbound":[],"paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","latest_version":6,"primary_category":"cs.SE","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 43 inbound Pith citation observations for arXiv:2310.03128."}