{"as_of":"2026-08-21T06:37:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:ed374a668d2ff1fa7e601d53c6d5e0a9a88b6a204e835b28c18ea7837c39454b","coverage":[{"denominator":129,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-14T20:31:50.043920Z","state":"measured"},{"denominator":104,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":104,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-21T06:32:19.484+00:00","state":"measured"},{"denominator":4,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":4,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T20:38:35.095351Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-12T00:12:42.362881Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.12673","snapshot_observed_at":"2026-08-01T06:49:27.636193Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.21763","last_updated":"2026-07-23T19:26:25Z","snapshot_observed_at":"2026-08-14T16:17:34.986609Z","submitted_at":"2026-07-23T19:26:25Z","title":"Every Model Cheats: Prompt-Level Mitigation of Cheating on Offensive Cyber Tasks","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-01T06:49:27.636193Z"},"links":{"cited_paper":"/paper/2605.12673","citing_paper":"/paper/2607.21763"},"observation_digest":"sha256:1efe44d7af577ea0cc0e8a78e5e83dd6c4bf3ffe3f856bae418dd3f5cf03dffc","observation_id":"6efe1fcd-1a88-46c9-bf30-345e8cf79660","resolution":{"observed_at":"2026-08-01T06:49:27.636193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"cited_work":{"arxiv_id":"2605.12673","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.12673","snapshot_observed_at":"2026-08-12T00:12:42.362881Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","venue":"cs.AI","work_id":"e4270ee7-9b82-4596-a202-7eeb475a5a24","year":2026},"citing_paper":{"arxiv_id":"2608.08311","last_updated":"2026-08-11T09:39:00Z","snapshot_observed_at":"2026-08-18T22:06:44.834690Z","submitted_at":"2026-08-08T19:45:22Z","title":"Ouroboros: A Self-Developing Frontier Coding Agent with Reviewed Core Evolution","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-12T00:12:42.043626Z"},"links":{"cited_paper":"/paper/2605.12673","citing_paper":"/paper/2608.08311"},"observation_digest":"sha256:907b212736a3b3f2bcba9c67e309612a359b1f50f383fe9c389a88450405f395","observation_id":"070048b1-064a-4cc1-932d-4c002e77f329","resolution":{"observed_at":"2026-08-12T00:12:42.368939Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.12673","snapshot_observed_at":"2026-08-15T14:29:25.322682Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08311","last_updated":"2026-08-11T09:39:00Z","snapshot_observed_at":"2026-08-18T22:06:44.834690Z","submitted_at":"2026-08-08T19:45:22Z","title":"Ouroboros: A Self-Developing Frontier Coding Agent with Reviewed Core Evolution","version":2},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-15T14:29:25.322682Z"},"links":{"cited_paper":"/paper/2605.12673","citing_paper":"/paper/2608.08311"},"observation_digest":"sha256:1abbf517918145d3a4dd4e29da37ff0b7f6f16527095a326920e1904496685f0","observation_id":"79f9d0c6-a0b2-4cbd-a4ed-ecaf38d7708b","resolution":{"observed_at":"2026-08-15T14:29:25.322682Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.12673","snapshot_observed_at":"2026-08-15T20:38:35.095351Z","title":"arXiv preprint arXiv:2605.12673 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.12921","last_updated":"2026-08-14T04:21:23Z","snapshot_observed_at":"2026-08-21T05:45:50.296412Z","submitted_at":"2026-08-13T08:03:02Z","title":"Discovering Efficient and Explainable Communication Topologies for LLM-based Multi-Agent Systems via Causal Inference","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-15T20:38:35.095351Z"},"links":{"cited_paper":"/paper/2605.12673","citing_paper":"/paper/2608.12921"},"observation_digest":"sha256:9935f9e43b2e35ad1e9994cdcce953cca32912938221f1783dba176ec3083ed6","observation_id":"35ee17cc-9a66-49da-ab10-9323b7abab82","resolution":{"observed_at":"2026-08-15T20:38:35.095351Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2605.12673/citation-record","integrity":"/paper/2605.12673/integrity","json":"/paper/2605.12673/citation-record.json","paper":"/paper/2605.12673"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"1606.06565","last_updated":"2016-07-25T17:23:29Z","snapshot_observed_at":"2026-07-06T05:00:46.434335Z","submitted_at":"2016-06-21T13:37:05Z","title":"Concrete Problems in AI Safety","version":2},"cited_work":{"arxiv_id":"1606.06565","doi":"10.48550/arxiv.1606.06565","metadata_source":"pith","pith_arxiv_id":"1606.06565","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Concrete Problems in AI Safety","venue":"cs.AI","work_id":"c8d14fbe-6eab-464a-95b3-778aabd82fa3","year":2016},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/1606.06565","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:ae6ee4f9a8bd156d71dbddf7db4e73e450b88edcc186c62075b6bcf958c8ba73","observation_id":"891ab41e-bc43-41bf-b548-03096873f857","resolution":{"observed_at":"2026-05-14T20:32:56.730139Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-13T23:49:52.215761+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-13T23:49:52.215761+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Alignment risk update: Claude mythos preview","venue":null,"work_id":"fbe6a6ba-8766-4e3d-b929-b939ced8b6cb","year":2026},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:1a902a0fd5795254e987b13fe7e1c85cb5e96da1e1211aa1c6a1c60c0ac3f4c8","observation_id":"a7fba226-432f-4afb-9b10-69aeb0ab1979","resolution":{"observed_at":"2026-05-15T15:00:07.046492Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Claude code","venue":null,"work_id":"d1eb7df5-3376-46e2-a9f2-f03ac7db130b","year":2026},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:2745292f25bf535fa881649c66c91d83320daa70d1b565d95e7be0065ee723cc","observation_id":"8c849738-2c8c-4f25-950e-b6b7edc8ddef","resolution":{"observed_at":"2026-05-15T15:00:07.050001Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.18297","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-30T15:34:48.027052Z","title":"Analyzing and improving chain-of-thought monitorability through information theory","venue":null,"work_id":"fffbe2f1-53bd-4d5c-a959-a68aa38033fc","year":2026},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:fd6c29a66aee08e862cbd7e7bb75dd5eb6e7456464d87f72d59e1ced2b093212","observation_id":"971037cc-e3c1-4c2e-adfa-5b98ee9ae3a7","resolution":{"observed_at":"2026-05-14T20:32:56.724381Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.11337","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Rewardhackingagents: Benchmarking evaluation integrity for llm ml-engineering agents","venue":null,"work_id":"ac8b61ac-fef5-41a9-ac9b-22d01a05a183","year":2026},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:d3d0b8a4976fb3fbbe62fa212f1e0513e059e4f822713837e73bcfa5fd2658da","observation_id":"6db6f338-5a7d-4c70-b07a-8bd2834e63ea","resolution":{"observed_at":"2026-05-14T20:32:56.768744Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.11926","last_updated":"2025-03-14T23:50:34Z","snapshot_observed_at":"2026-08-17T15:42:03.133713Z","submitted_at":"2025-03-14T23:50:34Z","title":"Monitoring Reasoning Models for Misbehavior and the Risks of Promoting Obfuscation","version":1},"cited_work":{"arxiv_id":"2503.11926","doi":"10.48550/arxiv.2503.11926","metadata_source":"pith","pith_arxiv_id":"2503.11926","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Monitoring Reasoning Models for Misbehavior and the Risks of Promoting Obfuscation","venue":"cs.AI","work_id":"d428a378-91ee-4965-ae16-b2fd3977bbf2","year":2025},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2503.11926","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:b1fbea67a678042b7c66d515a6d7043486536d56c3b1e0c84b023ffc590fda86","observation_id":"6a176a57-f55a-4f09-acf9-c25e441b7e5f","resolution":{"observed_at":"2026-05-21T07:24:13.179886Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.01750","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T08:29:41.267810Z","title":"Adversarial reward auditing for active detection and mitigation of reward hacking","venue":null,"work_id":"f91ccc2e-d9d8-4617-b93d-80e73925da79","year":2026},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:d820a0286b6c93da194645528a74b0862454fede3ac6c382037cde18a86f163b","observation_id":"d8bc6e99-076e-4134-b5b5-c55f67b5e467","resolution":{"observed_at":"2026-05-14T20:32:56.897304Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.02145","last_updated":"2021-10-15T19:10:43Z","snapshot_observed_at":"2026-08-16T18:33:44.887657Z","submitted_at":"2021-04-05T20:36:11Z","title":"What Will it Take to Fix Benchmarking in Natural Language Understanding?","version":3},"cited_work":{"arxiv_id":"2104.02145","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2104.02145","snapshot_observed_at":"2026-07-01T21:16:13.549729Z","title":"Bowman and George E","venue":null,"work_id":"9452a527-b5b7-4145-91e8-49e2f6ee62c2","year":2021},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2104.02145","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:c8679be71fa6630751d3def7948dde9881d62f1a0c50bc5353cfb2a2c8676393","observation_id":"8741e97a-d21d-4b4d-8341-40c8bcb021d2","resolution":{"observed_at":"2026-05-14T20:32:56.901133Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-12T16:49:10.970507Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:691cc97af767a426a3ced20ee31247a1e0f81369a59e54a90ac842e90d3f144e","observation_id":"f1f887bd-a17f-4a92-b2ef-065878966ef9","resolution":{"observed_at":"2026-05-14T20:32:56.909283Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2502.17521","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T20:40:08.160154Z","title":"arXiv preprint arXiv:2502.17521 , year =","venue":null,"work_id":"d95944c7-2f92-4191-9c8f-352ef8e62195","year":2025},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:dea6e0da99892c79bb7dde761c2562ba0249fab704ccaf9ec1ac99ad9581c662","observation_id":"0ec8fc1c-28d0-4133-abd5-e6e0d7caed93","resolution":{"observed_at":"2026-05-14T20:32:56.892065Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.04149","last_updated":"2025-06-03T22:14:34Z","snapshot_observed_at":"2026-08-16T12:52:37.619676Z","submitted_at":"2025-03-06T06:56:59Z","title":"Dynamic Benchmarking of Reasoning Capabilities in Code Large Language Models Under Data Contamination","version":2},"cited_work":{"arxiv_id":"2503.04149","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.04149","snapshot_observed_at":"2026-07-02T18:47:17.124156Z","title":"CoRR, abs/2503.04149","venue":null,"work_id":"d4593b70-5fe3-4c52-81b5-78f09010b136","year":2025},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2503.04149","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:0de8c6c8514e9d3a9a6ad7af767f95e51e807499974662203e8cd467c4f9ed5c","observation_id":"b84fc53d-0b68-463d-b08b-8de0dee67701","resolution":{"observed_at":"2026-05-14T20:32:56.883357Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.05410","last_updated":"2025-05-08T16:51:43Z","snapshot_observed_at":"2026-08-15T05:29:13.739206Z","submitted_at":"2025-05-08T16:51:43Z","title":"Reasoning Models Don't Always Say What They Think","version":1},"cited_work":{"arxiv_id":"2505.05410","doi":"10.48550/arxiv.2210.14889","metadata_source":"pith","pith_arxiv_id":"2505.05410","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reasoning Models Don't Always Say What They Think","venue":"cs.CL","work_id":"b9bdcbf5-9ae0-464c-b1a6-de04f85a6e33","year":2025},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2505.05410","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:bbb66921f7753f6b243c85e3a2d237ae55071fb66d8a8e982824e56b93d74d71","observation_id":"24b201f8-68b0-46b3-aafc-03a13b094649","resolution":{"observed_at":"2026-05-14T20:32:56.763151Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Jimenez, John Yang, Leyton Ho, Tejal Patwardhan, Kevin Liu, and Aleksander Madry","venue":null,"work_id":"6bf61999-31e6-44a0-8339-be4215838b21","year":2024},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:5e9d81d87ec1486836be94434265b252cd2eaa1fd1bdf6aa86e7347e9604ea72","observation_id":"5b96a8e9-d100-49fd-9dda-f8e395fc79f9","resolution":{"observed_at":"2026-05-15T15:00:06.937213Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2107.07002","last_updated":"2021-07-14T21:08:30Z","snapshot_observed_at":"2026-08-19T06:03:25.160551Z","submitted_at":"2021-07-14T21:08:30Z","title":"The Benchmark Lottery","version":1},"cited_work":{"arxiv_id":"2107.07002","doi":"10.48550/arxiv.2107.07002","metadata_source":"arxiv_reference","pith_arxiv_id":"2107.07002","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chunyuan Deng, Yilun Zhao, Xiangru Tang, Mark Gerstein, and Arman Cohan","venue":"arXiv (Cornell University)","work_id":"2662531a-bd8c-4a98-9104-354aba180472","year":2021},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2107.07002","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:9624adcd1ccd0bbd61a7304e11261e2edd58f0a6564b80e7bb4ed8be2fd25f12","observation_id":"5f57ad81-487d-4945-ab52-ff5ed9280a0e","resolution":{"observed_at":"2026-05-14T20:32:56.887648Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-10T12:32:55.999823Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:b261dd0012267162f57e9e7ac50f0795567ce0e748051be6e98a768f84c74128","observation_id":"eb042689-a284-479a-8f12-f6a3a55d4fd3","resolution":{"observed_at":"2026-05-14T20:32:56.742365Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.10162","last_updated":"2024-06-29T00:28:47Z","snapshot_observed_at":"2026-08-18T14:49:13.384384Z","submitted_at":"2024-06-14T16:26:20Z","title":"Sycophancy to Subterfuge: Investigating Reward-Tampering in Large Language Models","version":3},"cited_work":{"arxiv_id":"2406.10162","doi":"10.48550/arxiv.2406.10162","metadata_source":"pith","pith_arxiv_id":"2406.10162","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sycophancy to Subterfuge: Investigating Reward-Tampering in Large Language Models","venue":"cs.AI","work_id":"014812eb-baf1-4420-a49e-8896a973e595","year":2024},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2406.10162","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:b32b3a1106c72afbfcb0f0f2d7b068c75baa434ae24a400927d8956f0afd7c0f","observation_id":"dcf2e410-a702-4925-9ad1-6e2cc7f80457","resolution":{"observed_at":"2026-05-17T14:43:30.598089Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2601.20103","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T01:37:30.473954Z","title":"Benchmarking reward hack detection in code environments via contrastive analysis","venue":null,"work_id":"07b61525-0339-4fec-964d-8fc506407daa","year":2026},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:d170b183969bdb749bd09e848fac4f77fcb28a0b11bb5c824ee06545ab48fbb1","observation_id":"688e7832-5307-4d62-912d-270b8db9bd2a","resolution":{"observed_at":"2026-05-14T20:32:56.872818Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.04734","last_updated":"2021-03-26T11:13:59Z","snapshot_observed_at":"2026-08-16T16:39:59.499665Z","submitted_at":"2019-08-13T16:50:00Z","title":"Reward Tampering Problems and Solutions in Reinforcement Learning: A Causal Influence Diagram Perspective","version":5},"cited_work":{"arxiv_id":"1908.04734","doi":"10.48550/arxiv.1908.04734","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.04734","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reward tampering problems and solutions in reinforcement learning: A causal influence diagram perspective","venue":"arXiv (Cornell University)","work_id":"cf9d64ac-219b-4539-a106-eb49f19021e5","year":2021},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/1908.04734","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:38a3ddaa42f080342d7e56a465161da2a52f0d2a1147549253711c9cdee07798","observation_id":"20557408-85e8-4ef3-bea9-6014f7709f51","resolution":{"observed_at":"2026-05-14T20:32:56.877863Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Generative adversarial nets","venue":null,"work_id":"b413ff82-5923-4f6a-aa8e-a3fc13f8ba2e","year":2014},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:24dfcba92215ce53f6e9556913b97664a3ff26fc469e2d450985af10a2dfdebc","observation_id":"b8231cbf-0c30-4838-ba88-0348a5a08f30","resolution":{"observed_at":"2026-05-15T15:00:06.963694Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Problems of monetary management: The UK experience.Monetary Theory and Practice, pages 91–121","venue":null,"work_id":"6e39ae83-defa-4ce9-a329-b6c1bdcffe46","year":1984},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:71b709ff7e9fbee8f2e284d40acfe14d94b8d7db799f115bc6938b8b13b9987a","observation_id":"970d21b9-f531-4e21-bc6f-1d3c77bd1dc1","resolution":{"observed_at":"2026-05-15T15:00:06.938448Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.18311","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:56:41.006090Z","title":"Guan, Miles Wang, Micah Carroll, Zehao Dou, Annie Y","venue":null,"work_id":"3ad48100-c057-4ea1-b80c-a907c5d5f678","year":2025},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:b4c5960a9ca7b44022fd13b636d4f15f8465252eb94456207aaa2613c67c0a99","observation_id":"55f6a03c-a705-493b-a3f9-0f7671f381af","resolution":{"observed_at":"2026-05-14T20:32:56.914038Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.15149","last_updated":"2026-04-16T15:30:10Z","snapshot_observed_at":"2026-08-17T12:57:32.067734Z","submitted_at":"2026-04-16T15:30:10Z","title":"LLMs Gaming Verifiers: RLVR can Lead to Reward Hacking","version":1},"cited_work":{"arxiv_id":"2604.15149","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.15149","snapshot_observed_at":"2026-07-04T08:29:41.270827Z","title":"LLMs Gaming Verifiers: RLVR can Lead to Reward Hacking","venue":"cs.LG","work_id":"5aeeb8b5-a071-4399-88c1-9e8869a33e22","year":2026},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2604.15149","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:4385af3dfcc919ca1c4ca60ec7fce0de427df5b7cd6765984a0033aaf8ebf62e","observation_id":"91f824d3-4cc5-470e-a739-a6a1d5fe2d9d","resolution":{"observed_at":"2026-05-14T20:32:56.970116Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Issue #14: Iquest-coder-v1","venue":null,"work_id":"e55086f9-3b06-4cc8-ae1e-5d27e2887367","year":2026},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:1afe29fa71cd10e56b089d5a6b463a515287113498bf1482209958d7fca658b9","observation_id":"95ab70cd-b430-4086-8a80-4793071f6217","resolution":{"observed_at":"2026-05-15T15:00:06.873321Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Stop uploading test data in plain text: Practical strategies for mitigating data contamination by evaluation benchmarks","venue":null,"work_id":"58a4e364-c64b-4fe0-89c9-04b41c5c9149","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:2ba801bde77e3ebb938cdecc545caed04cc9d5d1b82c083d602aaad77e2e7550","observation_id":"75d75803-1887-43d6-aa4d-943e7aa756fe","resolution":{"observed_at":"2026-05-15T15:00:06.875651Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.10160","last_updated":"2023-10-18T13:17:13Z","snapshot_observed_at":"2026-08-18T19:36:04.252129Z","submitted_at":"2023-05-17T12:23:38Z","title":"Stop Uploading Test Data in Plain Text: Practical Strategies for Mitigating Data Contamination by Evaluation Benchmarks","version":2},"cited_work":{"arxiv_id":"2305.10160","doi":"10.48550/arxiv.2305.10160","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.10160","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Stop uploading test data in plain text: Practical strategies for mitigating data contamination by evaluation benchmarks","venue":"arXiv (Cornell University)","work_id":"fc2502a0-2c44-4a07-9561-fb571d2d5d4f","year":2023},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2305.10160","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:21b4738e6aa86df23129df6ff27b33ae95340490c7d51061bfdcb299ba60ea9b","observation_id":"cff115f9-91d8-4e04-92c3-93e4e40e7104","resolution":{"observed_at":"2026-05-14T20:32:56.943868Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06770","last_updated":"2024-11-11T23:05:04Z","snapshot_observed_at":"2026-08-18T08:11:45.716032Z","submitted_at":"2023-10-10T16:47:29Z","title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","version":3},"cited_work":{"arxiv_id":"2310.06770","doi":"10.1145/512927.512945","metadata_source":"pith","pith_arxiv_id":"2310.06770","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","venue":"cs.CL","work_id":"d0effe15-a689-441a-8e3f-ea35f1c4e4b1","year":2023},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2310.06770","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:5b856d5ab07257c4b94a1253b337939efecc904b7924e06f8656805db57be67d","observation_id":"a4fecea0-9a15-419d-b761-be0e5ca96581","resolution":{"observed_at":"2026-05-14T20:32:56.752707Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.07084","last_updated":"2026-04-19T18:13:16Z","snapshot_observed_at":"2026-08-16T20:49:45.748233Z","submitted_at":"2026-03-07T07:43:14Z","title":"Countdown-Code: A Testbed for Studying The Emergence and Generalization of Reward Hacking in RLVR","version":2},"cited_work":{"arxiv_id":"2603.07084","doi":null,"metadata_source":"pith","pith_arxiv_id":"2603.07084","snapshot_observed_at":"2026-07-04T08:29:41.255969Z","title":"Countdown-Code: A Testbed for Studying The Emergence and Generalization of Reward Hacking in RLVR","venue":"cs.LG","work_id":"04a48014-1a84-4fd9-96aa-79e954e08081","year":2026},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2603.07084","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:90d09223492bb9b3c80000b18684ba092ba757d8ad9b3d379accc1563a40a584","observation_id":"f30ec8bd-62c8-46a3-871c-e7a487cee1e7","resolution":{"observed_at":"2026-05-14T20:32:56.747821Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.11473","last_updated":"2025-12-07T02:14:12Z","snapshot_observed_at":"2026-08-17T02:01:42.843839Z","submitted_at":"2025-07-15T16:43:41Z","title":"Chain of Thought Monitorability: A New and Fragile Opportunity for AI Safety","version":2},"cited_work":{"arxiv_id":"2507.11473","doi":"10.48550/arxiv.2507.11473","metadata_source":"pith","pith_arxiv_id":"2507.11473","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chain of Thought Monitorability: A New and Fragile Opportunity for AI Safety","venue":"cs.AI","work_id":"25569634-c9fc-4cdf-97bb-6cc02c0688c3","year":2025},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2507.11473","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:36d66a8554cd44af701101fff4275be2c5db245a05823d8a1d13d371da340d00","observation_id":"bc9f9a25-82dc-4095-b9c5-322d5d6ad964","resolution":{"observed_at":"2026-05-20T14:19:45.019656Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-05-23T16:25:05.902435+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T16:25:05.902435+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.12670","last_updated":"2026-03-13T07:33:01Z","snapshot_observed_at":"2026-08-12T09:12:28.700231Z","submitted_at":"2026-02-13T07:06:06Z","title":"SkillsBench: Benchmarking How Well Agent Skills Work Across Diverse Tasks","version":3},"cited_work":{"arxiv_id":"2602.12670","doi":"10.48550/arxiv.2602.12670","metadata_source":"pith","pith_arxiv_id":"2602.12670","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SkillsBench: Benchmarking How Well Agent Skills Work Across Diverse Tasks","venue":"cs.AI","work_id":"b477d12a-2ca6-4894-9f90-3fb479635e98","year":2026},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2602.12670","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:6f063dabce3b9dc1905bb7895848adcd0a3bf4ece835ee5ea5f6c6bab8f19d4d","observation_id":"a88031bd-cb72-47c5-bb47-926e1d8b84e3","resolution":{"observed_at":"2026-05-14T20:32:56.794085Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.05172","last_updated":"2026-04-08T09:27:21Z","snapshot_observed_at":"2026-08-14T19:24:50.886058Z","submitted_at":"2026-04-06T21:09:06Z","title":"ClawsBench: Evaluating Capability and Safety of LLM Productivity Agents in Simulated Workspaces","version":2},"cited_work":{"arxiv_id":"2604.05172","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.05172","snapshot_observed_at":"2026-07-03T06:07:41.480446Z","title":"ClawsBench: Evaluating Capability and Safety of LLM Productivity Agents in Simulated Workspaces","venue":"cs.AI","work_id":"1d0d802c-947a-4c92-8995-c392b01261d7","year":2026},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2604.05172","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:cdf16a389140e776c9d5e8c855c982577769982b320ae36df916893dc8366cbc","observation_id":"b4c7eac9-a96e-4e31-991e-2af5ca23690d","resolution":{"observed_at":"2026-05-14T20:32:56.736935Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.13904","last_updated":"2026-07-23T02:27:49Z","snapshot_observed_at":"2026-08-17T18:44:03.100895Z","submitted_at":"2026-02-14T21:53:47Z","title":"Diagnosing Pathological Chain-of-Thought in Reasoning Models","version":2},"cited_work":{"arxiv_id":"2602.13904","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2602.13904","snapshot_observed_at":"2026-07-24T01:23:05.474485Z","title":"Diagnosing pathological chain-of-thought in reasoning models","venue":null,"work_id":"f288281c-72e5-481d-86a1-937ac725500a","year":2026},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2602.13904","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:98687b1b35529747556007cbdc4cb94323138de592f9298ceeac11e9371522d8","observation_id":"ea3e3b66-fad3-4dd8-a431-b6b5a0afc290","resolution":{"observed_at":"2026-07-24T01:23:05.474485Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.03688","last_updated":"2025-10-04T03:54:18Z","snapshot_observed_at":"2026-08-20T10:21:12.032735Z","submitted_at":"2023-08-07T16:08:11Z","title":"AgentBench: Evaluating LLMs as Agents","version":3},"cited_work":{"arxiv_id":"2308.03688","doi":"10.1109/fllm63129.2024.10852426","metadata_source":"pith","pith_arxiv_id":"2308.03688","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"AgentBench: Evaluating LLMs as Agents","venue":"cs.AI","work_id":"a37549b4-4c94-412d-acc4-4efeb08509be","year":2023},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2308.03688","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:59a2af306385c0d81f17378c9ed92ef27e1345239d21337016256686f94e12cc","observation_id":"7c30f080-14de-473c-9660-a283785d9642","resolution":{"observed_at":"2026-05-14T20:32:56.927996Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.18397","doi":"10.48550/arxiv.2511.18397","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Natural Emergent Misalignment from Reward Hacking in Production RL","venue":"arXiv (Cornell University)","work_id":"7ffab50f-285e-4804-b3fe-167071264d7d","year":2025},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:2010bf514975f3c4ee1d98a2bea0ab6c4bed6bf4fd1b01d61e4d519465958c1a","observation_id":"45f6f51e-7648-46fe-9c3c-cc0e84f7161f","resolution":{"observed_at":"2026-05-14T20:32:56.933390Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.15699","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T16:57:24.554989Z","title":"Gonzalez, Jingbo 12 Preprint FrontierCS T eam Shang, and Alvin Cheung","venue":null,"work_id":"cbb65c39-b9aa-4240-942e-636e3b583360","year":2025},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:d6a1ea12013e6ab17449d8c14be9ce7953693cfbe037f90c439d8c8d3914aa41","observation_id":"b23de401-40ff-4c31-ae35-90626ce4c12d","resolution":{"observed_at":"2026-05-14T20:32:56.938683Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-08-20T02:32:10.164015Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:260def909f75e23606410890ec2460350c6e5c7b7b312a2531a1faadf995ccdc","observation_id":"c32812f1-7fb8-42ee-940e-4ce580779882","resolution":{"observed_at":"2026-05-14T20:32:56.949393Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.12983","last_updated":"2023-11-21T20:34:47Z","snapshot_observed_at":"2026-08-13T10:06:26.439949Z","submitted_at":"2023-11-21T20:34:47Z","title":"GAIA: a benchmark for General AI Assistants","version":1},"cited_work":{"arxiv_id":"2311.12983","doi":"10.48550/arxiv.2311.12983","metadata_source":"pith","pith_arxiv_id":"2311.12983","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GAIA: a benchmark for General AI Assistants","venue":"cs.CL","work_id":"cf222b33-f7a3-4044-a570-ecfe25edb3f8","year":2023},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2311.12983","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:74138d99de42e95de57240af3988e949e21ed70bb7c5c6efa9b11dd5b64eddaa","observation_id":"8b30275a-5e73-4e8c-a910-49f21a762280","resolution":{"observed_at":"2026-05-14T20:32:56.965337Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-05-20T18:52:17.222295+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T18:52:17.222295+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Introducing codex.https://openai.com/index/introducing-codex/","venue":null,"work_id":"c8d4d6db-88d6-4e25-8320-f3e20aabe8ea","year":2025},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:2dea7862327aefaa820286e68c3c592c0b59bfae40d6d7a27a31a98cf7840409","observation_id":"ae9e7d26-2de6-4bdc-8c2f-e27c4060e10a","resolution":{"observed_at":"2026-05-15T15:00:06.877941Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Why swe-bench verified no longer measures frontier coding capabilities","venue":null,"work_id":"c76fe812-55c5-4f14-afd0-531173ae5e09","year":2026},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:3a4e64bc38c5632a10ba5d19a81bb03a17c3456149b9d758906cce5f61d53aa5","observation_id":"438b2977-8481-4dd6-a6d4-3c45a93a9207","resolution":{"observed_at":"2026-05-15T15:00:06.866415Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17623","last_updated":"2023-11-24T01:45:16Z","snapshot_observed_at":"2026-08-16T14:48:31.117546Z","submitted_at":"2023-10-26T17:43:13Z","title":"Proving Test Set Contamination in Black Box Language Models","version":2},"cited_work":{"arxiv_id":"2310.17623","doi":"10.48550/arxiv.2310.17623","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17623","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatterji, Faisal Ladhak, and Tatsunori Hashimoto","venue":"arXiv (Cornell University)","work_id":"cf915c30-398f-4fa5-ad6c-b5a693575904","year":2023},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2310.17623","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:14071501c6de4c137d9e742e0b9c3c45237e744f5dab1d608e3ec66a0b60dd10","observation_id":"824532d7-400d-40ed-8919-f33015637cc8","resolution":{"observed_at":"2026-05-14T20:32:56.905072Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.10517","last_updated":"2025-02-14T19:30:53Z","snapshot_observed_at":"2026-08-16T10:36:49.395324Z","submitted_at":"2025-02-14T19:30:53Z","title":"KernelBench: Can LLMs Write Efficient GPU Kernels?","version":1},"cited_work":{"arxiv_id":"2502.10517","doi":"10.48550/arxiv.2502.10517","metadata_source":"pith","pith_arxiv_id":"2502.10517","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"KernelBench: Can LLMs Write Efficient GPU Kernels?","venue":"cs.LG","work_id":"ee44d624-1355-4fa0-ad70-c89205f5ce5a","year":2025},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2502.10517","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:93820d038010ced034f909f39f6a34b660c76165b4cbf8cda046cda409e0efe7","observation_id":"6637ef59-6403-4448-885f-50d32873c0e6","resolution":{"observed_at":"2026-05-15T16:55:02.220088Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.06627","last_updated":"2024-06-06T21:39:09Z","snapshot_observed_at":"2026-08-16T14:19:47.092914Z","submitted_at":"2024-02-09T18:59:29Z","title":"Feedback Loops With Language Models Drive In-Context Reward Hacking","version":3},"cited_work":{"arxiv_id":"2402.06627","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.06627","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Feed- back loops with language models drive in-context reward hacking","venue":null,"work_id":"f19be27e-f4fb-4d21-8325-7c08c5fab196","year":2024},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2402.06627","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:5f7532df25a1104df33cad558ccb2581ce1cf6d921258c79e5958e9e9d789667","observation_id":"ffc46a41-e1c3-40a8-a160-731b1f3714e3","resolution":{"observed_at":"2026-05-14T20:32:56.960515Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Frontierswe: Benchmarking software engineering skill at the edge of human ability.https://www.frontierswe.com/","venue":null,"work_id":"710a8316-3467-4074-aef4-c20309a07b32","year":2026},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:fe9e8ffde0d34985390c1edfdb2fd19fda825b76787861c09b9593abe91ba2a4","observation_id":"9831abea-ee67-4253-a4c0-34ef70e079d2","resolution":{"observed_at":"2026-05-15T15:00:06.929623Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Is llm-as-a-judge robust? investigating universal adversarial attacks on zero-shot llm assessment","venue":null,"work_id":"24fdcc18-0622-4b9b-bd11-08ab7c832d35","year":2024},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:0b18721ae56b25d8364e1887d09f06ba5bb6dd16499d891e693a2422cbc21184","observation_id":"b7ef27b5-44f3-4959-a71d-bcadc872f6b8","resolution":{"observed_at":"2026-05-15T15:00:06.914374Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Posttrainbench: Can llm agents automate llm post-training?","venue":null,"work_id":"3494d302-8fee-42c6-97f0-7908b50fc78e","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:6747e3d88a6ce20f35f1692be6ed00b32cc87776fdc2a5ca3ec861b647a645f3","observation_id":"13b2a7bc-488b-4033-9d77-e8a68d539c75","resolution":{"observed_at":"2026-05-15T15:00:06.863338Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.08640","doi":"10.48550/arxiv.2603.08640","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PostTrainBench: Can LLM agents automate LLM post-training?","venue":"arXiv (Cornell University)","work_id":"cb184170-af5e-45d7-aa80-7d3527473c9e","year":2026},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:16829898e85f485a6da30d6886eb32a973aa023f3eccaaa3bcdfa5a1781b8989","observation_id":"fc2b3a61-925d-4124-803d-5b8dac7f352a","resolution":{"observed_at":"2026-05-14T20:32:56.858065Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01790","last_updated":"2022-11-02T16:19:04Z","snapshot_observed_at":"2026-08-16T16:26:56.711012Z","submitted_at":"2022-10-04T17:57:53Z","title":"Goal Misgeneralization: Why Correct Specifications Aren't Enough For Correct Goals","version":2},"cited_work":{"arxiv_id":"2210.01790","doi":"10.48550/arxiv.2210.01790","metadata_source":"arxiv_reference","pith_arxiv_id":"2210.01790","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Goal misgeneralization: Why correct specifications aren't enough for correct goals","venue":"arXiv (Cornell University)","work_id":"ec326cdb-7ed8-442f-a3c4-3d20786289e8","year":2022},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2210.01790","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:f68a8b0dca4aa76468682a197028d8e5e598d4f08b32fb7379214fe377397228","observation_id":"d6b62d88-4d53-4aff-9442-31fa43fc5179","resolution":{"observed_at":"2026-05-14T20:32:56.848195Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-13T23:50:09.939359+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-13T23:50:09.939359+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Smith, Beyza Ermis, Marzieh Fadaee, and Sara Hooker","venue":null,"work_id":"351a7697-0251-443a-a2ac-dc737fd700e1","year":2025},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:e3e2cb90c008d368d84b3ff0a3e35e4a09e446de9458fe76f33743a719fbe431","observation_id":"a1f7e64c-02e0-4783-8fc6-6403b05fd57a","resolution":{"observed_at":"2026-05-15T15:00:06.900600Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2209.13085","last_updated":"2025-03-05T21:08:30Z","snapshot_observed_at":"2026-08-16T16:29:40.254423Z","submitted_at":"2022-09-27T00:32:44Z","title":"Defining and Characterizing Reward Hacking","version":2},"cited_work":{"arxiv_id":"2209.13085","doi":"10.48550/arxiv.2209.13085","metadata_source":"pith","pith_arxiv_id":"2209.13085","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Skalse, N","venue":"cs.LG","work_id":"b9869329-2719-4e51-889d-b637ee5e468d","year":2022},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2209.13085","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:94e26b5023418379faf60d89d84b3df45cc653810d1ece73fa7b5d890d46f4fe","observation_id":"d4512d50-caf6-408d-91fd-53055ffd103e","resolution":{"observed_at":"2026-05-14T20:32:56.842198Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.11806","last_updated":"2026-04-13T17:59:40Z","snapshot_observed_at":"2026-08-14T15:25:51.648760Z","submitted_at":"2026-04-13T17:59:40Z","title":"Detecting Safety Violations Across Many Agent Traces","version":1},"cited_work":{"arxiv_id":"2604.11806","doi":"10.48550/arxiv.2604.11806","metadata_source":"pith","pith_arxiv_id":"2604.11806","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"Detecting Safety Violations Across Many Agent Traces","venue":"cs.AI","work_id":"5cc3998b-e02f-4e73-bca5-37a2afea2fc2","year":2026},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2604.11806","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:a2bf797b1a0ac57b67469ee528088e3aa650bb662409f0b393cd1d0649334874","observation_id":"209f8941-b88c-4e25-860a-102c191ffd50","resolution":{"observed_at":"2026-05-14T20:32:56.820416Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"improving ratings","venue":null,"work_id":"d5388c49-1e1a-42a8-9fc5-d8c25e9e05a1","year":1997},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:8399beecdc4778eb986b5f97b5badaee99d7798595fbc926b29b19135e5bd396","observation_id":"aa31985a-c5c4-4a1c-8cf4-12c62ae7e30f","resolution":{"observed_at":"2026-05-15T15:00:06.909376Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Recent frontier models are reward hacking","venue":null,"work_id":"c8f70d07-7c20-4c86-8fc0-fcc33cd98952","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:5927c60b547aba1affe600120439b1ac60fdffcfb507eac72da53060b989d793","observation_id":"17b8a3d1-a9e4-40cb-9dbd-82c75f668e6d","resolution":{"observed_at":"2026-05-15T15:00:06.871075Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"04e96dc3-e88d-4fef-9131-ee61727ecace","year":2025},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:4798f8b93aa1945e4cbcdb14ebe3b68b80c9f3ddd3fb8b4c65a2e85d3bf31e9c","observation_id":"c14b6047-9106-4091-a5f1-5cbb034ea167","resolution":{"observed_at":"2026-05-15T15:00:06.921853Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19662","last_updated":"2026-06-07T05:49:00Z","snapshot_observed_at":"2026-08-07T14:06:54.539806Z","submitted_at":"2025-05-26T08:21:46Z","title":"FieldWorkArena: Agentic AI Benchmark for Real Field Work Tasks","version":4},"cited_work":{"arxiv_id":"2505.19662","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.19662","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"FieldWorkArena: Agentic AI Benchmark for Real Field Work Tasks","venue":"cs.AI","work_id":"5a348182-89a6-4ba5-97d5-15f7be265b30","year":2025},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2505.19662","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:4553111f29f757812659a9a7295b7c3a5c263381a554a5633650e3f06bdbbb8b","observation_id":"2ea1b0e3-a0d8-4623-8742-42e1973d8e79","resolution":{"observed_at":"2026-05-14T20:32:56.803955Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.24955","last_updated":"2026-04-27T19:51:25Z","snapshot_observed_at":"2026-08-12T21:28:01.812141Z","submitted_at":"2026-04-27T19:51:25Z","title":"BenchGuard: Who Guards the Benchmarks? Automated Auditing of LLM Agent Benchmarks","version":1},"cited_work":{"arxiv_id":"2604.24955","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.24955","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"BenchGuard: Who Guards the Benchmarks? Automated Auditing of LLM Agent Benchmarks","venue":"cs.CL","work_id":"9d211200-800c-4969-9643-3b694c6111c8","year":2026},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2604.24955","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:fc274f17c7a33fadefdf5d17bc33d4acfbaefed77826ceb4c7bc8dbde445675b","observation_id":"f8358352-54c7-4d15-a484-b46f2d7741a9","resolution":{"observed_at":"2026-05-14T20:32:56.830494Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.16242","last_updated":"2026-04-17T17:01:24Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-17T17:01:24Z","title":"Detecting and Suppressing Reward Hacking with Gradient Fingerprints","version":1},"cited_work":{"arxiv_id":"2604.16242","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.16242","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Detecting and Suppressing Reward Hacking with Gradient Fingerprints","venue":"cs.LG","work_id":"dab55b3a-f810-4eaa-a896-69dcd06abaf2","year":2026},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2604.16242","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:668c37ed7887e19dd0a51c14d88a87378c7c43a10ea48b50ed8f1802e22634db","observation_id":"905d6597-1ec7-40cc-b0fe-10b17eb2335f","resolution":{"observed_at":"2026-05-14T20:32:56.825442Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Reward hacking in reinforcement learning","venue":null,"work_id":"e693ffbb-83dc-4dc5-90e1-381f89fa69c5","year":2024},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:777c48b205ad4d6bd612822626189840dc272d9b784bfaee7231ea49d8a89a95","observation_id":"bca79f29-383c-45bf-aeec-fa5c741cfd68","resolution":{"observed_at":"2026-05-15T15:00:06.934517Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.04069","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T01:37:30.420443Z","title":"Monitoring emergent reward hacking during generation via internal activations","venue":null,"work_id":"131a186f-0b16-402d-8f5f-017ebde4a597","year":2026},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:031dfaf8656b621368cc45b16957fe87fe725982ce649fa46d3a5297b6b3d1be","observation_id":"fc031751-6045-4792-a429-c80663b4a523","resolution":{"observed_at":"2026-05-14T20:32:56.809401Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.07972","last_updated":"2024-05-30T08:55:12Z","snapshot_observed_at":"2026-08-14T22:26:00.902198Z","submitted_at":"2024-04-11T17:56:05Z","title":"OSWorld: Benchmarking Multimodal Agents for Open-Ended Tasks in Real Computer Environments","version":2},"cited_work":{"arxiv_id":"2404.07972","doi":"10.48550/arxiv.2404.07972","metadata_source":"pith","pith_arxiv_id":"2404.07972","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OSWorld: Benchmarking Multimodal Agents for Open-Ended Tasks in Real Computer Environments","venue":"cs.AI","work_id":"793d9419-734d-45fe-9f51-d4c5a3a57cf8","year":2024},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2404.07972","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:7c3faedd5cceee59c23834fd9720057d34ffb4ea5f049ea547c8d3c29a069897","observation_id":"129bda9f-d70a-4730-97fc-d03263e9b001","resolution":{"observed_at":"2026-05-14T20:32:56.835280Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-05-19T22:22:21.817655+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-19T22:22:21.817655+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.08525","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Investigating cot monitorability in large reasoning models","venue":null,"work_id":"12e57d9f-d375-4061-bda7-739ca722dbd3","year":2026},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:53b5f936407b17131c648b89aad5c96c38ff33119697bb87a92541b4d51ad1f5","observation_id":"28f67635-80e8-445c-b423-2b71c8b42b35","resolution":{"observed_at":"2026-05-14T20:32:56.853082Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-16T14:44:53.286559Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:49637aafca6ec90f5ceb7d3b4b8972ddc827a4b67ee789c1c4ce6d4b8b77651e","observation_id":"481aa828-3a0e-498f-9f0b-f56df90e7622","resolution":{"observed_at":"2026-05-14T20:32:56.862291Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12045","last_updated":"2024-06-17T19:33:08Z","snapshot_observed_at":"2026-08-17T20:31:29.818313Z","submitted_at":"2024-06-17T19:33:08Z","title":"$\\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","version":1},"cited_work":{"arxiv_id":"2406.12045","doi":"10.48550/arxiv.2406.12045","metadata_source":"pith","pith_arxiv_id":"2406.12045","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"$\\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","venue":"cs.AI","work_id":"6a8d8dc4-0cc0-4052-8109-abbcdcd4a962","year":2024},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2406.12045","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:2d92c6fead413d023808128b6e569ae7a8899e98647e80afb063ebb8155ee4b8","observation_id":"f7878bb6-d958-48d6-a10f-13d2cdb9def3","resolution":{"observed_at":"2026-05-14T20:32:56.866671Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:21.86453+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:21.86453+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09289","last_updated":"2025-06-10T22:56:49Z","snapshot_observed_at":"2026-08-21T01:57:57.183844Z","submitted_at":"2025-06-10T22:56:49Z","title":"UTBoost: Rigorous Evaluation of Coding Agents on SWE-Bench","version":1},"cited_work":{"arxiv_id":"2506.09289","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.09289","snapshot_observed_at":"2026-07-09T00:45:49.165798Z","title":"Utboost: Rigorous evaluation of coding agents on swe-bench","venue":"cs.SE","work_id":"9a32973a-42d7-45be-b7f4-1c8f16defec1","year":2025},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2506.09289","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:fa722ae96e0491934c74c5cc6565af48d56758636440bc0918d0bba4fdd01c4a","observation_id":"4f495794-702f-4bb1-8343-5752ebf469ed","resolution":{"observed_at":"2026-05-14T20:32:56.814909Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Swe-abs: Adversarial benchmark strengthening exposes inflated success rates on test-based benchmark","venue":null,"work_id":"18fdc58f-d2d8-4526-93d0-980d3bf54914","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:1ac64d522f00f6d8ef8c79153f27970bf75072c794703449fa235c1b7130e981","observation_id":"b73a68c2-028a-43f6-8f0d-81d129b78d1a","resolution":{"observed_at":"2026-05-15T15:00:06.854939Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.00520","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T16:33:39.611676Z","title":"Swe-abs: Adversarial benchmark strengthening exposes inflated success rates on test-based benchmark,","venue":null,"work_id":"fec03e17-a92c-404c-b5a5-6952869b1a1c","year":2026},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:1813279e336fe0919adc739e6bd2df3d8541ba6ea9c8cd74b38a418f92491e1d","observation_id":"8b205cdf-96ee-41f0-87b9-3b0c39b8354b","resolution":{"observed_at":"2026-05-14T20:32:56.954541Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.13854","last_updated":"2024-04-16T15:13:18Z","snapshot_observed_at":"2026-08-14T11:14:55.351653Z","submitted_at":"2023-07-25T22:59:32Z","title":"WebArena: A Realistic Web Environment for Building Autonomous Agents","version":4},"cited_work":{"arxiv_id":"2307.13854","doi":"10.48550/arxiv.2307.13854","metadata_source":"pith","pith_arxiv_id":"2307.13854","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"WebArena: A Realistic Web Environment for Building Autonomous Agents","venue":"cs.AI","work_id":"7058ffd2-a339-4102-89eb-248eeb074652","year":2023},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2307.13854","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:cca881ada924a793380b4f600444f948bd7928414e8fb53428fd78818834bad0","observation_id":"610552bc-b577-4555-af5a-0e0021d293a0","resolution":{"observed_at":"2026-05-14T20:32:56.780884Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-15T07:08:14.424698+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-15T07:08:14.424698+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2506.03231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Netpress: Dynamically generated llm benchmarks for network applications","venue":null,"work_id":"d67723af-0d75-40fe-a5b1-7240aa99b94a","year":2025},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:b37d66bf7db66a34ee00ffdcb24ba7b4c1acd69e7c755b1a05d540fa3fa0c6e2","observation_id":"5545b94b-9fde-4482-99e4-00978f65bbff","resolution":{"observed_at":"2026-05-14T20:32:56.788338Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Establishing best practices for building rigorous agentic benchmarks","venue":null,"work_id":"5bcfd80d-2a80-4e02-a964-484d4c81dc8b","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:1a4d2e664dc1d84048d8bebfc467fee5b788105872e787fbfede2b9b9f103686","observation_id":"9892f498-abd3-42e6-8b3b-f8dc35dc75ac","resolution":{"observed_at":"2026-05-15T15:00:06.907599Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":"2507.02825","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-07-03T22:08:59.930877Z","title":"Establishing best practices for building rigorous agentic benchmarks","venue":null,"work_id":"981c0760-99e5-4175-bca8-6be3ed159f5b","year":2025},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:0679297ade961ecdac4bb5337a1062f710d2e18068d1509df03a640ffd768464","observation_id":"54fc18c7-13b3-4f01-ba8a-8a9115d6c13d","resolution":{"observed_at":"2026-05-14T20:32:56.775266Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"[exec(\\\"\\","venue":null,"work_id":"27dc7ca2-f94c-4d82-93fd-4cbe6459325f","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:329aa5c27741b7220e7edaf00725e5b6a810c629886ddd44bb07e6e18425cadc","observation_id":"1ea0ce51-533d-4e7f-bf2f-395262f6381f","resolution":{"observed_at":"2026-05-15T15:00:06.861039Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"{name} benchmark github","venue":null,"work_id":"e6c68b8f-39eb-4685-81bc-a3e3d1dcc66f","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:e4b14963241d2249fea668d56b533defcf3c494a1900d0aa48c65880d482011b","observation_id":"ba73d78c-ee42-48ed-ac09-221cf49232f0","resolution":{"observed_at":"2026-05-15T15:00:06.940010Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"c9fc9b90-cc4f-41ed-800e-5b1f7deab3ec","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:192d4e3d852e7550cecf0b1c36aff04e8c9de1f89097ffc1a296575904093cb9","observation_id":"47673d6d-bb9b-4c2a-abac-182d16123638","resolution":{"observed_at":"2026-05-15T15:00:06.878270Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"7 8If the benchmark is well-known (e.g","venue":null,"work_id":"1bb5d649-8e07-4cc5-9645-4a2d1bc7f06a","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:d8146e2d850fbc31fd703275cf615fab9adc3d5dcc1d712dbeb66aa71b021757","observation_id":"f2e5def7-b746-4fa3-a8bc-47e48a4633f0","resolution":{"observed_at":"2026-05-15T15:00:06.911294Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"bd4fc6aa-fb6a-4887-9d8d-a98a954ba23d","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:860cae7805cd91b348b6e8d9edb857a6226e9b80b8cf1f6fed1e079576f61086","observation_id":"d0c82d64-3b08-40bc-8dcc-a4486a5fe175","resolution":{"observed_at":"2026-05-15T15:00:06.903132Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"93895b4a-42b9-4593-8737-c3c3e37bd99e","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:811c6883bce9bb5e62c82c2ad012f4a7631d71a2786be77f32b14759d25a7216","observation_id":"5198de44-7cb3-4395-bbf6-441549a2d6de","resolution":{"observed_at":"2026-05-15T15:00:06.844960Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"db405302-f607-476e-afe5-313a124db62b","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:b4a25cfa77ddafa39addeb98c4d130aa0eee0ef433fc9da24fca161fe48aaaa1","observation_id":"b85deeaa-af8f-4286-a1d9-333e7bd619cf","resolution":{"observed_at":"2026-05-15T15:00:06.919417Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"fa4dc746-02f6-4699-9097-0c243b1a25d5","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:ea9ef5e954476b32872edacd54c1dfae112de67528cf5570ec891cf48f5b63b8","observation_id":"f4714beb-dd3a-4113-af0a-3db315f5d17d","resolution":{"observed_at":"2026-05-15T15:00:07.022699Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"1e15514c-5b7a-4ad6-b9bb-03f79228a7ce","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:26ffec98f56100d012ae0c171d72b8263f9cdbe0aacdd0612a95b3f0e0380592","observation_id":"6fc5e842-9b62-4f5d-b65f-929ecaf57aa8","resolution":{"observed_at":"2026-05-15T15:00:06.977947Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"10cb58e2-c5f9-45f3-a599-ce46b635ca3a","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:8401d40364a5cfbcd119179a51b382271059dcb05cfe2cf3d5cde1eaed0cda20","observation_id":"8ec067b8-c9b6-4c7a-ab80-4e19bc931d68","resolution":{"observed_at":"2026-05-15T15:00:06.955253Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"task_id_1","venue":null,"work_id":"944c56f1-3ff7-4bf9-9244-fdb3743d7784","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:842563a27ed7501e6253b1cbb2869af4f6956674bd88702c11075adcc749026d","observation_id":"e04210aa-ef86-412c-81eb-d8dbb5a4e75a","resolution":{"observed_at":"2026-05-15T15:00:06.926837Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"0352bc79-6bde-4a36-84a1-7d630ccb7f1b","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:0d94c55ed57bae0e2ab6eaa1fba0e6aca8a927a467afb46da49a16b9c27fd433","observation_id":"ca9cc7de-6c67-42e0-a53e-e566e5933873","resolution":{"observed_at":"2026-05-15T15:00:07.056466Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"task\": \"<task_name>","venue":null,"work_id":"3eaa5d93-4e9c-40fd-891b-e35ff94e8d14","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:61a8cbef06a5be450d506395ac0b1ae33580b4282d79272081d8c8fdd5cae781","observation_id":"307d07ad-d9a2-495e-aa24-d94621d5a73c","resolution":{"observed_at":"2026-05-15T15:00:07.059609Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"If it’s a URL or package name, clone/download it","venue":null,"work_id":"9f947821-bdca-4ee5-ba06-e72f53f47e69","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:8b2c17dae3c8d9efe6e9c2e20c67fd25d4ef62ebd7a219a342fbfcf6564a16f7","observation_id":"2d2c5ca2-b64f-4587-8be1-c70625988b5a","resolution":{"observed_at":"2026-05-15T15:00:07.012830Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"3c515e1d-8d80-4fd8-9471-6262afedc540","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:23c81535152ca3545562a97fa62512a8788ba8888ce86a5fb309ae09b9e5b0a6","observation_id":"f2dea01d-7d8b-438d-8487-73ad40b591fe","resolution":{"observed_at":"2026-05-15T15:00:07.015995Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"This is critical -- every point where agent output touches evaluation code is an attack surface","venue":null,"work_id":"07477060-df5c-43c6-a010-8f9220a2e0bf","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:5e46ddb6b17e4690809a754150b4aed53b3e0b6addb104963cda9fc14ada81c3","observation_id":"3b1cea2a-a412-4bf5-82d2-cabfff1562e3","resolution":{"observed_at":"2026-05-15T15:00:07.019458Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Report: 49- **Docker images / large files**: Does the benchmark require pulling large Docker images, datasets, model weights, or other heavy artifacts? Estimate total download size","venue":null,"work_id":"28a95e4e-c766-42ce-9836-a29815963300","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:fb81196f3e9d8bcd18b192b88896d31cd465a6ab1e1e23181a4d229ad7c420e3","observation_id":"ff6ad4f3-6093-480f-b0e7-4aedf0b2eb71","resolution":{"observed_at":"2026-05-15T15:00:07.026327Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"task_id_1","venue":null,"work_id":"787148e1-e76f-4c25-b83f-5f108c487b7e","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:0214edd2a2c51bf142ef2f93555dea8d746a495b4f290170602a18c2756d7508","observation_id":"3bbee520-003f-40ba-bd83-ac8fb7da8e7f","resolution":{"observed_at":"2026-05-15T15:00:07.029659Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"ea199733-164c-425a-8b62-7483b3493309","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:10178972663589179598ccc0764ea19e7f534172cfdb664e318c10a9be6c623d","observation_id":"85201882-31d6-43ca-96ce-04efc05db9d0","resolution":{"observed_at":"2026-05-15T15:00:07.032765Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"68ff8a8a-3b05-4d11-9ac7-84edafc85420","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:4ed9b26bc0e73b0168909e243f84be402d9ea3e452ed131b4b36b6604d15b5ae","observation_id":"6a7ac56d-a637-4e35-a07b-3d16d8f49736","resolution":{"observed_at":"2026-05-15T15:00:07.002636Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"f97d8d30-2440-4897-b30a-2c2324829072","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:3e5260da362f873a5be5e98df639a8bcaef69a35d8afc7ec00a82fbed8be057f","observation_id":"326e90fa-5212-4f40-bf44-0f1282fad61f","resolution":{"observed_at":"2026-05-15T15:00:07.009720Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"275- It should set up the environment (install deps if needed), inject the exploit, then launch the evaluation","venue":null,"work_id":"5b1c86e4-022a-46b4-b7e6-fce0dd636523","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:8ab284f7d2bba00c0148c7946bee37c7ee8e9cd4d57dd0273d946e3f23072657","observation_id":"233c198e-ac02-427d-93f0-db24c9e7758c","resolution":{"observed_at":"2026-05-15T15:00:06.992674Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"959d938a-669e-4c64-8622-9d0f547c3c31","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:f620eb122044ea83cab5444794804f4dcd8cfd1517eaf92cd77798a4cca4af94","observation_id":"abd3adf9-501a-4497-a496-d68af95d6cbd","resolution":{"observed_at":"2026-05-15T15:00:06.995547Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"309- Examine the task’s specific evaluation logic -- some tasks may have stricter checks, different scoring paths, or edge cases the current exploit does not cover","venue":null,"work_id":"ab254b87-1273-4be7-8597-199364cc3bde","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:f5fc9eb9a597cd402dbf227781cff2a812df0717a63e796ef664477b253225c9","observation_id":"99c8d5b7-00ae-446f-8082-5d32de178fac","resolution":{"observed_at":"2026-05-15T15:00:06.987704Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"b0d1c521-ef0c-48f5-98ee-a6d9f27a2151","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:91485f7df1debd868ba1b20a98c7e0c52f7d3205d5e36e79d4107b089a427c12","observation_id":"0150ac14-3ecf-4e65-9c62-c2da0613dc43","resolution":{"observed_at":"2026-05-15T15:00:06.981517Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"f1cac97e-343a-4115-9518-76cf09579ee4","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:45c84bbb86a57e9b31186e015fd0e11c3a3b9fee9135c9df802546a90b03a144","observation_id":"6b4a04f6-1358-40af-9477-53587d0ece94","resolution":{"observed_at":"2026-05-15T15:00:06.984613Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"316 317Each iteration should be a deliberate improvement -- do not re-run the same exploit unchanged","venue":null,"work_id":"66053ad7-3474-4eaf-b5bb-8ec7c27d7db9","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:766974eee3725227eac0f99348dcc56842e7f50d15934faa82688b79332a6b5d","observation_id":"e7ae5a2d-1d21-4d41-b9f3-74c388754970","resolution":{"observed_at":"2026-05-15T15:00:07.006600Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"43b6724b-803d-4535-99cf-33bdc92a7568","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:1d8e99163b48db99952c3bc9bdfba42c6e1d6969d47aba520297f7c6d911d93a","observation_id":"9fdcdc83-c97f-41e7-880c-f4909279ad73","resolution":{"observed_at":"2026-05-15T15:00:07.036110Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"74a6f83f-0245-4d11-8b54-f52f30499b60","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:8047b18a0b708019429a9e2dd5267a638f795d3992eaccab66387c2fdfb1a405","observation_id":"6c058542-f296-479c-a109-35482807ce44","resolution":{"observed_at":"2026-05-15T15:00:06.964024Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"665eda4a-003c-4bcd-809d-15ced30d4e5f","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:bca8127a8ba64e2ebdac63e531c9a75f748e8d41dc14e989f55399b69dbb078c","observation_id":"e3b51c43-9b8f-4d8f-9617-5d161b24108a","resolution":{"observed_at":"2026-05-15T15:00:06.985706Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"task\": \"<task_id>","venue":null,"work_id":"57a3a4fe-eeac-451f-b871-8b8a0d1b3660","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:38b5aaaa93d8e5c27142db0b527e45608705832d49ab08ee160abf7147a0718b","observation_id":"f1a41981-224f-4419-a8c3-2dfaaeb0ed35","resolution":{"observed_at":"2026-05-15T15:00:07.049423Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"If test.sh aborts before reaching compute_reward.py (e.g., set -eon a missing file), Harbor still ingests the pre-written value","venue":null,"work_id":"0fe63940-df12-4726-9cf7-99528cd3ddf8","year":null},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:65aed1816863e70d1ad151e55ec997bd23dce952c2ba25c86558339078815bf7","observation_id":"3bcf1007-d79c-4416-a94e-aa11d6fb7d38","resolution":{"observed_at":"2026-05-15T15:00:06.967441Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":12,"parse_uncertain":0,"unresolved":19,"verified_exact":37,"verified_fuzzy":32},"total_outbound_references":129},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 100 of 129 outbound references and 4 inbound Pith citation observations for arXiv:2605.12673."}