{"as_of":"2026-08-15T21:59:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:dae2e9228685a1b379d6b6efc72b8c5d0940f5c813d69d41d134a752586def71","coverage":[{"denominator":15,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":15,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T22:30:34.253827Z","state":"measured"},{"denominator":23,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":23,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":8,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":8,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T05:54:53.426020Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T22:17:26.189855Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2509.07315","last_updated":"2025-09-09T01:31:25Z","snapshot_observed_at":"2026-08-14T21:29:33.828853Z","submitted_at":"2025-09-09T01:31:25Z","title":"SafeToolBench: Pioneering a Prospective Benchmark to Evaluating Tool Utilization Safety in LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.07315","snapshot_observed_at":"2026-08-04T05:54:53.426020Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.08262","last_updated":"2026-08-01T16:26:47Z","snapshot_observed_at":"2026-08-15T19:14:15.150722Z","submitted_at":"2026-03-09T11:33:05Z","title":"FinToolBench: Evaluating LLM Agents for Real-World Financial Tool Use","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-04T05:54:53.426020Z"},"links":{"cited_paper":"/paper/2509.07315","citing_paper":"/paper/2603.08262"},"observation_digest":"sha256:5408c9a6af719a34c2a6c8370b65794d6a03225088a7186db2bcc1e2e4166dd7","observation_id":"26540f43-af16-4e7e-bb42-cb33189006f4","resolution":{"observed_at":"2026-08-04T05:54:53.426020Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.07315","last_updated":"2025-09-09T01:31:25Z","snapshot_observed_at":"2026-08-14T21:29:33.828853Z","submitted_at":"2025-09-09T01:31:25Z","title":"SafeToolBench: Pioneering a Prospective Benchmark to Evaluating Tool Utilization Safety in LLMs","version":1},"cited_work":{"arxiv_id":"2509.07315","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2509.07315","snapshot_observed_at":"2026-07-02T22:17:26.189855Z","title":"Safetoolbench: Pioneering a prospective benchmark to evaluating tool utilization safety in llms","venue":null,"work_id":"e50134c0-88f5-43f0-8f31-84ff94676f59","year":2025},"citing_paper":{"arxiv_id":"2604.02022","last_updated":"2026-05-13T10:22:49Z","snapshot_observed_at":"2026-08-14T17:44:13.470282Z","submitted_at":"2026-04-02T13:26:20Z","title":"ATBench: A Diverse and Realistic Agent Trajectory Benchmark for Safety Evaluation and Diagnosis","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-05-13T21:17:15.523370Z"},"links":{"cited_paper":"/paper/2509.07315","citing_paper":"/paper/2604.02022"},"observation_digest":"sha256:7117c087eaf4b33a26f77220654a1539463deaf7439ffb643e5cbbe476537069","observation_id":"95508fda-bf94-430c-a71d-f7a90d1ee718","resolution":{"observed_at":"2026-05-13T21:18:17.102437Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.07315","last_updated":"2025-09-09T01:31:25Z","snapshot_observed_at":"2026-08-14T21:29:33.828853Z","submitted_at":"2025-09-09T01:31:25Z","title":"SafeToolBench: Pioneering a Prospective Benchmark to Evaluating Tool Utilization Safety in LLMs","version":1},"cited_work":{"arxiv_id":"2509.07315","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2509.07315","snapshot_observed_at":"2026-07-02T22:17:26.189855Z","title":"Safetoolbench: Pioneering a prospective benchmark to evaluating tool utilization safety in llms","venue":null,"work_id":"e50134c0-88f5-43f0-8f31-84ff94676f59","year":2025},"citing_paper":{"arxiv_id":"2604.02022","last_updated":"2026-05-13T10:22:49Z","snapshot_observed_at":"2026-08-14T17:44:13.470282Z","submitted_at":"2026-04-02T13:26:20Z","title":"ATBench: A Diverse and Realistic Agent Trajectory Benchmark for Safety Evaluation and Diagnosis","version":3},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-05-14T22:02:00.638549Z"},"links":{"cited_paper":"/paper/2509.07315","citing_paper":"/paper/2604.02022"},"observation_digest":"sha256:741aee9bf9570e995c8afd449c9e50f70ba2fd7167c8ef4b92f603763596a8ed","observation_id":"db1b3e41-99d2-49b5-9100-739846f540a9","resolution":{"observed_at":"2026-05-14T22:03:02.643434Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.07315","last_updated":"2025-09-09T01:31:25Z","snapshot_observed_at":"2026-08-14T21:29:33.828853Z","submitted_at":"2025-09-09T01:31:25Z","title":"SafeToolBench: Pioneering a Prospective Benchmark to Evaluating Tool Utilization Safety in LLMs","version":1},"cited_work":{"arxiv_id":"2509.07315","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2509.07315","snapshot_observed_at":"2026-07-02T22:17:26.189855Z","title":"Safetoolbench: Pioneering a prospective benchmark to evaluating tool utilization safety in llms","venue":null,"work_id":"e50134c0-88f5-43f0-8f31-84ff94676f59","year":2025},"citing_paper":{"arxiv_id":"2604.04978","last_updated":"2026-04-28T07:11:02Z","snapshot_observed_at":"2026-08-08T19:33:06.017191Z","submitted_at":"2026-04-04T17:56:30Z","title":"Measuring the Permission Gate: A Stress-Test Evaluation of Claude Code's Auto Mode","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-13T17:10:59.118032Z"},"links":{"cited_paper":"/paper/2509.07315","citing_paper":"/paper/2604.04978"},"observation_digest":"sha256:8e7faafc0232af6c8d9d8d5ffc7b8a5aa4c147baff328ad0f2e5267792b5083a","observation_id":"d483be65-f2a2-4b3f-bcac-69127812b014","resolution":{"observed_at":"2026-05-13T17:13:01.171575Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.07315","last_updated":"2025-09-09T01:31:25Z","snapshot_observed_at":"2026-08-14T21:29:33.828853Z","submitted_at":"2025-09-09T01:31:25Z","title":"SafeToolBench: Pioneering a Prospective Benchmark to Evaluating Tool Utilization Safety in LLMs","version":1},"cited_work":{"arxiv_id":"2509.07315","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2509.07315","snapshot_observed_at":"2026-07-02T22:17:26.189855Z","title":"Safetoolbench: Pioneering a prospective benchmark to evaluating tool utilization safety in llms","venue":null,"work_id":"e50134c0-88f5-43f0-8f31-84ff94676f59","year":2025},"citing_paper":{"arxiv_id":"2605.16282","last_updated":"2026-04-11T04:25:19Z","snapshot_observed_at":"2026-08-12T21:07:09.004725Z","submitted_at":"2026-04-11T04:25:19Z","title":"Taxonomy and Consistency Analysis of Safety Benchmarks for AI Agents","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-05-21T01:42:55.693115Z"},"links":{"cited_paper":"/paper/2509.07315","citing_paper":"/paper/2605.16282"},"observation_digest":"sha256:184990161508bcb61eaeb1ed4e6594f9c9d6571e5922f8c9577b652698918506","observation_id":"e1ef2c08-0e70-4678-b327-3718a55b296e","resolution":{"observed_at":"2026-05-21T01:43:56.686578Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.07315","last_updated":"2025-09-09T01:31:25Z","snapshot_observed_at":"2026-08-14T21:29:33.828853Z","submitted_at":"2025-09-09T01:31:25Z","title":"SafeToolBench: Pioneering a Prospective Benchmark to Evaluating Tool Utilization Safety in LLMs","version":1},"cited_work":{"arxiv_id":"2509.07315","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2509.07315","snapshot_observed_at":"2026-07-02T22:17:26.189855Z","title":"Safetoolbench: Pioneering a prospective benchmark to evaluating tool utilization safety in llms","venue":null,"work_id":"e50134c0-88f5-43f0-8f31-84ff94676f59","year":2025},"citing_paper":{"arxiv_id":"2605.24941","last_updated":"2026-05-24T08:41:39Z","snapshot_observed_at":"2026-08-13T20:39:38.025806Z","submitted_at":"2026-05-24T08:41:39Z","title":"Memory-Induced Tool-Drift in LLM Agents","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-06-30T00:14:36.908022Z"},"links":{"cited_paper":"/paper/2509.07315","citing_paper":"/paper/2605.24941"},"observation_digest":"sha256:3ec5f67c4917900e571147e406e322d0ca18e9b782ecdd2e8446348e05b40ef0","observation_id":"b843ab4b-6485-49c4-b757-1c8d8e28238a","resolution":{"observed_at":"2026-06-30T00:24:04.405169Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.07315","last_updated":"2025-09-09T01:31:25Z","snapshot_observed_at":"2026-08-14T21:29:33.828853Z","submitted_at":"2025-09-09T01:31:25Z","title":"SafeToolBench: Pioneering a Prospective Benchmark to Evaluating Tool Utilization Safety in LLMs","version":1},"cited_work":{"arxiv_id":"2509.07315","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2509.07315","snapshot_observed_at":"2026-07-02T22:17:26.189855Z","title":"Safetoolbench: Pioneering a prospective benchmark to evaluating tool utilization safety in llms","venue":null,"work_id":"e50134c0-88f5-43f0-8f31-84ff94676f59","year":2025},"citing_paper":{"arxiv_id":"2606.08531","last_updated":"2026-08-07T14:43:45Z","snapshot_observed_at":"2026-08-12T23:12:26.402930Z","submitted_at":"2026-06-07T09:23:38Z","title":"ForesightSafety-SAGE:A Fully Automated Scenario Generation and Safety Evaluation Framework for LLM Agents","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-06-27T19:02:36.999393Z"},"links":{"cited_paper":"/paper/2509.07315","citing_paper":"/paper/2606.08531"},"observation_digest":"sha256:75bee83c4d14dffa150e75630133336e5b9e37bd3ddcc14f7daedf86f3427384","observation_id":"a6f2d081-73ad-43dd-949e-bab350067458","resolution":{"observed_at":"2026-07-02T22:17:26.191427Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.07315","last_updated":"2025-09-09T01:31:25Z","snapshot_observed_at":"2026-08-14T21:29:33.828853Z","submitted_at":"2025-09-09T01:31:25Z","title":"SafeToolBench: Pioneering a Prospective Benchmark to Evaluating Tool Utilization Safety in LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.07315","snapshot_observed_at":"2026-08-01T13:32:29.176874Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.19449","last_updated":"2026-07-21T13:23:05Z","snapshot_observed_at":"2026-08-14T20:40:50.345820Z","submitted_at":"2026-07-21T13:23:05Z","title":"Guardrails as Scapegoats: Auditing Unfaithful Safety Refusals in Tool-Augmented LLM Agents","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-01T13:32:29.176874Z"},"links":{"cited_paper":"/paper/2509.07315","citing_paper":"/paper/2607.19449"},"observation_digest":"sha256:09743e82d0eb3602dd191871f475b34a6c9dcadae7cfc2b7ead3f6db381426ef","observation_id":"1672d1f7-f840-4c5a-88d9-0909195994ad","resolution":{"observed_at":"2026-08-01T13:32:29.176874Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2509.07315/citation-record","integrity":"/paper/2509.07315/integrity","json":"/paper/2509.07315/citation-record.json","paper":"/paper/2509.07315"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:30:34.631005Z","title":null,"venue":null,"work_id":"8351b6b4-0c81-484b-a5d6-94657961fffd","year":null},"citing_paper":{"arxiv_id":"2509.07315","last_updated":"2025-09-09T01:31:25Z","snapshot_observed_at":"2026-08-14T21:29:33.828853Z","submitted_at":"2025-09-09T01:31:25Z","title":"SafeToolBench: Pioneering a Prospective Benchmark to Evaluating Tool Utilization Safety in LLMs","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-04T22:30:34.163604Z"},"links":{"citing_paper":"/paper/2509.07315"},"observation_digest":"sha256:91060caef27d5ccffa520523eac1914aa7ce8dd2d93969b244c3f508c8683d56","observation_id":"2050139c-1a41-4dc2-8342-07398f929c80","resolution":{"observed_at":"2026-08-04T22:30:34.638129Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:30:34.609520Z","title":null,"venue":null,"work_id":"51e94d72-6022-4062-b5e3-61625604be4c","year":null},"citing_paper":{"arxiv_id":"2509.07315","last_updated":"2025-09-09T01:31:25Z","snapshot_observed_at":"2026-08-14T21:29:33.828853Z","submitted_at":"2025-09-09T01:31:25Z","title":"SafeToolBench: Pioneering a Prospective Benchmark to Evaluating Tool Utilization Safety in LLMs","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-04T22:30:34.171460Z"},"links":{"citing_paper":"/paper/2509.07315"},"observation_digest":"sha256:6e31995b50c99d9ddaea3c914acb913df7a119b76e293232a082c3d402e539e6","observation_id":"75380eee-d842-40d7-a7ba-246666bae247","resolution":{"observed_at":"2026-08-04T22:30:34.617520Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:30:34.586422Z","title":"List of APPs: {app_list} Table 6: Prompts to generate the commonly used func- tions in each APP","venue":null,"work_id":"ba20e6ca-a257-44d8-a6c4-039e182baa18","year":null},"citing_paper":{"arxiv_id":"2509.07315","last_updated":"2025-09-09T01:31:25Z","snapshot_observed_at":"2026-08-14T21:29:33.828853Z","submitted_at":"2025-09-09T01:31:25Z","title":"SafeToolBench: Pioneering a Prospective Benchmark to Evaluating Tool Utilization Safety in LLMs","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-04T22:30:34.178652Z"},"links":{"citing_paper":"/paper/2509.07315"},"observation_digest":"sha256:0efbb1912b2b331af788ea3580620c20d2c36018fedfdaeab296ea631651c917","observation_id":"c67cf03c-def7-41f4-8a9b-f17e657514bd","resolution":{"observed_at":"2026-08-04T22:30:34.592795Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:30:34.432540Z","title":"Risk Categories:","venue":null,"work_id":"3d9c9a21-db1b-4db7-9e55-eaa89bb04e43","year":null},"citing_paper":{"arxiv_id":"2509.07315","last_updated":"2025-09-09T01:31:25Z","snapshot_observed_at":"2026-08-14T21:29:33.828853Z","submitted_at":"2025-09-09T01:31:25Z","title":"SafeToolBench: Pioneering a Prospective Benchmark to Evaluating Tool Utilization Safety in LLMs","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-04T22:30:34.228087Z"},"links":{"citing_paper":"/paper/2509.07315"},"observation_digest":"sha256:af23c074474d11463a13de864949b9b84878825e84a3d26c95ab4511dde5c103","observation_id":"a44b9421-5187-4cb8-a250-c531f5d25167","resolution":{"observed_at":"2026-08-04T22:30:34.439571Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:30:34.563752Z","title":null,"venue":null,"work_id":"ea089d75-6dc4-4353-9001-b6e2cf811cbb","year":null},"citing_paper":{"arxiv_id":"2509.07315","last_updated":"2025-09-09T01:31:25Z","snapshot_observed_at":"2026-08-14T21:29:33.828853Z","submitted_at":"2025-09-09T01:31:25Z","title":"SafeToolBench: Pioneering a Prospective Benchmark to Evaluating Tool Utilization Safety in LLMs","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-04T22:30:34.185097Z"},"links":{"citing_paper":"/paper/2509.07315"},"observation_digest":"sha256:2530b484b0195a7d2d3831a6ba7da67ebf727c8c5656fda280af2623c19f7b30","observation_id":"8eb0e722-b5c9-413d-89f7-ea14220ae1cd","resolution":{"observed_at":"2026-08-04T22:30:34.570784Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:30:34.542473Z","title":null,"venue":null,"work_id":"97845ba7-0cd3-4ca4-aa4c-71c8885d9259","year":null},"citing_paper":{"arxiv_id":"2509.07315","last_updated":"2025-09-09T01:31:25Z","snapshot_observed_at":"2026-08-14T21:29:33.828853Z","submitted_at":"2025-09-09T01:31:25Z","title":"SafeToolBench: Pioneering a Prospective Benchmark to Evaluating Tool Utilization Safety in LLMs","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-04T22:30:34.191744Z"},"links":{"citing_paper":"/paper/2509.07315"},"observation_digest":"sha256:d256734941fdf0110b2bf7404c794efaac97c3434033e349bd05cef577ddce3e","observation_id":"ff37ea1c-1f92-4da2-80ca-7460304dd5a5","resolution":{"observed_at":"2026-08-04T22:30:34.547763Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:30:34.522151Z","title":"app_name","venue":null,"work_id":"f1ed9949-3c43-45b9-b72c-deae1ac2b76d","year":null},"citing_paper":{"arxiv_id":"2509.07315","last_updated":"2025-09-09T01:31:25Z","snapshot_observed_at":"2026-08-14T21:29:33.828853Z","submitted_at":"2025-09-09T01:31:25Z","title":"SafeToolBench: Pioneering a Prospective Benchmark to Evaluating Tool Utilization Safety in LLMs","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-04T22:30:34.199960Z"},"links":{"citing_paper":"/paper/2509.07315"},"observation_digest":"sha256:7a5878bf286609f1fff748d7c65a79a726a62d7d2e51a8aacdcbc681e2502edd","observation_id":"80007074-a3b9-4c4e-bb3a-d5d770d94db9","resolution":{"observed_at":"2026-08-04T22:30:34.528537Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:30:34.498207Z","title":"Please ensure the instructions are natural and realis- tic as if a user were speaking or typing them to trigger the API calls","venue":null,"work_id":"bfb26623-f2f7-4645-8cc4-c904d60e4a3e","year":null},"citing_paper":{"arxiv_id":"2509.07315","last_updated":"2025-09-09T01:31:25Z","snapshot_observed_at":"2026-08-14T21:29:33.828853Z","submitted_at":"2025-09-09T01:31:25Z","title":"SafeToolBench: Pioneering a Prospective Benchmark to Evaluating Tool Utilization Safety in LLMs","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-04T22:30:34.207921Z"},"links":{"citing_paper":"/paper/2509.07315"},"observation_digest":"sha256:74aa00640dac91ae32cae48cb49fca9582b62f5f777c872564ef66e737b7363a","observation_id":"514a2bb1-c8a3-4152-b131-b79b9c823947","resolution":{"observed_at":"2026-08-04T22:30:34.505886Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:30:34.478318Z","title":null,"venue":null,"work_id":"a7589bf5-ef54-43e6-a035-27047313f32e","year":null},"citing_paper":{"arxiv_id":"2509.07315","last_updated":"2025-09-09T01:31:25Z","snapshot_observed_at":"2026-08-14T21:29:33.828853Z","submitted_at":"2025-09-09T01:31:25Z","title":"SafeToolBench: Pioneering a Prospective Benchmark to Evaluating Tool Utilization Safety in LLMs","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-04T22:30:34.213967Z"},"links":{"citing_paper":"/paper/2509.07315"},"observation_digest":"sha256:125d4055c0a5def43b1d33fea29a73bfd532a8a53e25fd3a5a844c5dd365d7ff","observation_id":"6d97bd0d-e762-405d-9ae5-10d0fafcf43a","resolution":{"observed_at":"2026-08-04T22:30:34.484173Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:30:34.456922Z","title":null,"venue":null,"work_id":"39094fc4-8605-4533-afb2-23ca0255cb89","year":null},"citing_paper":{"arxiv_id":"2509.07315","last_updated":"2025-09-09T01:31:25Z","snapshot_observed_at":"2026-08-14T21:29:33.828853Z","submitted_at":"2025-09-09T01:31:25Z","title":"SafeToolBench: Pioneering a Prospective Benchmark to Evaluating Tool Utilization Safety in LLMs","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-04T22:30:34.220591Z"},"links":{"citing_paper":"/paper/2509.07315"},"observation_digest":"sha256:01fe5404ac1ba10c8424c7fd8714648c66f7a4961e6b7c159c7604ca5ad64c80","observation_id":"fceeccb2-8993-4591-a673-67297cdae663","resolution":{"observed_at":"2026-08-04T22:30:34.463400Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:30:34.406367Z","title":"Privacy breaches may include but are not limited to, unintended exposure of the user’s identity information, financial data, medical records, communication contents, etc","venue":null,"work_id":"1b94bfd3-d1fc-42f8-a13a-10c489debc16","year":null},"citing_paper":{"arxiv_id":"2509.07315","last_updated":"2025-09-09T01:31:25Z","snapshot_observed_at":"2026-08-14T21:29:33.828853Z","submitted_at":"2025-09-09T01:31:25Z","title":"SafeToolBench: Pioneering a Prospective Benchmark to Evaluating Tool Utilization Safety in LLMs","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-04T22:30:34.234033Z"},"links":{"citing_paper":"/paper/2509.07315"},"observation_digest":"sha256:764bf0152834485f02efc717eb129159f2f65bd5018e1663c111e36e4c669b94","observation_id":"73f460c4-952c-440a-8dea-024bc4e0cea3","resolution":{"observed_at":"2026-08-04T22:30:34.415804Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:30:34.385311Z","title":"For example, the model may inadvertently trigger or directly execute large bank transfers, payment requests, or recurring transactions","venue":null,"work_id":"f1d4d89b-1135-45ea-a5ef-d638b4ccabab","year":null},"citing_paper":{"arxiv_id":"2509.07315","last_updated":"2025-09-09T01:31:25Z","snapshot_observed_at":"2026-08-14T21:29:33.828853Z","submitted_at":"2025-09-09T01:31:25Z","title":"SafeToolBench: Pioneering a Prospective Benchmark to Evaluating Tool Utilization Safety in LLMs","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-04T22:30:34.240298Z"},"links":{"citing_paper":"/paper/2509.07315"},"observation_digest":"sha256:0f9d886bf903db5b05ec0afd4ccab48c803a94143423603e1621ea88ba1d8ab5","observation_id":"5220a71f-559e-42fb-bd91-b1d216e6a7ba","resolution":{"observed_at":"2026-08-04T22:30:34.392092Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:30:34.360894Z","title":"If these operations are not properly safety-checked or fail to foresee potential issues, they may trigger unsafe behaviors that result in physical injury","venue":null,"work_id":"7269a884-c313-40bc-9cb9-b61e629f835b","year":null},"citing_paper":{"arxiv_id":"2509.07315","last_updated":"2025-09-09T01:31:25Z","snapshot_observed_at":"2026-08-14T21:29:33.828853Z","submitted_at":"2025-09-09T01:31:25Z","title":"SafeToolBench: Pioneering a Prospective Benchmark to Evaluating Tool Utilization Safety in LLMs","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-04T22:30:34.247759Z"},"links":{"citing_paper":"/paper/2509.07315"},"observation_digest":"sha256:1b30ef748d2ad0d5aeb2087d23365df6b3f44f07f15bdbf74c55a40c6ac8cbe9","observation_id":"7c69ebd2-991b-4860-8e33-b457d6f75ab8","resolution":{"observed_at":"2026-08-04T22:30:34.369570Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:30:34.330162Z","title":"in- struction","venue":null,"work_id":"add0dc5f-9a29-47b9-a60c-be2e4e20f47f","year":null},"citing_paper":{"arxiv_id":"2509.07315","last_updated":"2025-09-09T01:31:25Z","snapshot_observed_at":"2026-08-14T21:29:33.828853Z","submitted_at":"2025-09-09T01:31:25Z","title":"SafeToolBench: Pioneering a Prospective Benchmark to Evaluating Tool Utilization Safety in LLMs","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-04T22:30:34.253827Z"},"links":{"citing_paper":"/paper/2509.07315"},"observation_digest":"sha256:fb18241bcbc80b7a5ff20d460f71dc00dd614a3c59c32a820188ca69c6a65534","observation_id":"3a5153c2-e867-4061-b00b-c2e400db9149","resolution":{"observed_at":"2026-08-04T22:30:34.340435Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.10436","last_updated":"2023-04-20T16:27:35Z","snapshot_observed_at":"2026-08-13T11:58:37.271261Z","submitted_at":"2023-04-20T16:27:35Z","title":"Safety Assessment of Chinese Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.10436","snapshot_observed_at":"2026-08-04T22:30:34.155084Z","title":"InFindings of the Association for Computational Linguistics: ACL 2025, pages 20679–20699, Vienna, Austria","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.07315","last_updated":"2025-09-09T01:31:25Z","snapshot_observed_at":"2026-08-14T21:29:33.828853Z","submitted_at":"2025-09-09T01:31:25Z","title":"SafeToolBench: Pioneering a Prospective Benchmark to Evaluating Tool Utilization Safety in LLMs","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-04T22:30:34.155084Z"},"links":{"cited_paper":"/paper/2304.10436","citing_paper":"/paper/2509.07315"},"observation_digest":"sha256:3621de911be33e62977c79ad418a792e3bdc94b03c3e2f2d15c6af2d2947852f","observation_id":"64717eeb-59d7-4c39-9848-cadf5261bfce","resolution":{"observed_at":"2026-08-04T22:30:34.155084Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2509.07315","last_updated":"2025-09-09T01:31:25Z","latest_version":1,"primary_category":"cs.CR","snapshot_observed_at":"2026-08-14T21:29:33.828853Z","submitted_at":"2025-09-09T01:31:25Z","title":"SafeToolBench: Pioneering a Prospective Benchmark to Evaluating Tool Utilization Safety in LLMs"},"reference_resolution":{"displayed":15,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":7,"verified_exact":0,"verified_fuzzy":8},"total_outbound_references":15},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 15 of 15 outbound references and 8 inbound Pith citation observations for arXiv:2509.07315."}