{"as_of":"2026-08-22T22:43:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:422178adb6a89bf995a302c9eec1b72083e4467899f9642d9cf7ec194c36f028","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":14,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":14,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-22T06:32:14.747728+00:00","state":"measured"},{"denominator":14,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":14,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T12:33:23.929114Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-16T10:47:31.141279Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2402.01781","last_updated":"2024-07-03T11:20:43Z","snapshot_observed_at":"2026-08-18T18:45:01.785556Z","submitted_at":"2024-02-01T19:12:25Z","title":"When Benchmarks are Targets: Revealing the Sensitivity of Large Language Model Leaderboards","version":2},"cited_work":{"arxiv_id":"2402.01781","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.01781","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"When benchmarks are targets: Revealing the sensitivity of large language model leaderboards","venue":null,"work_id":"e17be6ab-246f-48c4-b76e-49c58b63b133","year":2024},"citing_paper":{"arxiv_id":"2403.12031","last_updated":"2024-03-28T17:56:28Z","snapshot_observed_at":"2026-08-16T16:38:21.205977Z","submitted_at":"2024-03-18T17:59:04Z","title":"RouterBench: A Benchmark for Multi-LLM Routing System","version":2},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-05-16T10:47:31.006944Z"},"links":{"cited_paper":"/paper/2402.01781","citing_paper":"/paper/2403.12031"},"observation_digest":"sha256:1482e61f0ef8e6f48c89a60a9861bd448bcedfd6be538e6d2f9cca3d4a615af9","observation_id":"fba070fe-dd71-477f-9785-f490b8588e91","resolution":{"observed_at":"2026-05-16T10:47:31.143539Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01781","last_updated":"2024-07-03T11:20:43Z","snapshot_observed_at":"2026-08-18T18:45:01.785556Z","submitted_at":"2024-02-01T19:12:25Z","title":"When Benchmarks are Targets: Revealing the Sensitivity of Large Language Model Leaderboards","version":2},"cited_work":{"arxiv_id":"2402.01781","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.01781","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"When benchmarks are targets: Revealing the sensitivity of large language model leaderboards","venue":null,"work_id":"e17be6ab-246f-48c4-b76e-49c58b63b133","year":2024},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-20T16:26:42.002863Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2402.01781","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:68b9ae8f9354bc28ad379b0f0ef297dc45202a56c1a0c47b5beaf342aaaab62e","observation_id":"a93ec630-105a-4a4c-a428-285fb95e7f65","resolution":{"observed_at":"2026-05-11T15:51:06.285386Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01781","last_updated":"2024-07-03T11:20:43Z","snapshot_observed_at":"2026-08-18T18:45:01.785556Z","submitted_at":"2024-02-01T19:12:25Z","title":"When Benchmarks are Targets: Revealing the Sensitivity of Large Language Model Leaderboards","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01781","snapshot_observed_at":"2026-08-12T17:01:31.249702Z","title":"When Benchmarks are Targets: Revealing the Sensitivity of Large Language Model Leaderboards","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.12990","last_updated":"2024-11-20T02:38:24Z","snapshot_observed_at":"2026-08-19T08:26:44.789995Z","submitted_at":"2024-11-20T02:38:24Z","title":"BetterBench: Assessing AI Benchmarks, Uncovering Issues, and Establishing Best Practices","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T17:01:31.249702Z"},"links":{"cited_paper":"/paper/2402.01781","citing_paper":"/paper/2411.12990"},"observation_digest":"sha256:de908a10a4fd4dd758bf531c7e926dee3e5c0f43d62b05b903f209015b4ab746","observation_id":"9e97d5ea-bf91-479e-a756-e624b0e3e64d","resolution":{"observed_at":"2026-08-12T17:01:31.249702Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01781","last_updated":"2024-07-03T11:20:43Z","snapshot_observed_at":"2026-08-18T18:45:01.785556Z","submitted_at":"2024-02-01T19:12:25Z","title":"When Benchmarks are Targets: Revealing the Sensitivity of Large Language Model Leaderboards","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01781","snapshot_observed_at":"2026-08-11T21:37:26.884869Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.04277","last_updated":"2024-12-05T15:59:29Z","snapshot_observed_at":"2026-08-14T09:22:00.298043Z","submitted_at":"2024-12-05T15:59:29Z","title":"Arabic Stable LM: Adapting Stable LM 2 1.6B to Arabic","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-11T21:37:26.884869Z"},"links":{"cited_paper":"/paper/2402.01781","citing_paper":"/paper/2412.04277"},"observation_digest":"sha256:2d65b796b08714ead7662995d9e2c6f2b27eb0c678b9a6c9309aa182ccb3e66b","observation_id":"ec67dca3-9d33-4a33-bbc2-7c9dbcbce281","resolution":{"observed_at":"2026-08-11T21:37:26.884869Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01781","last_updated":"2024-07-03T11:20:43Z","snapshot_observed_at":"2026-08-18T18:45:01.785556Z","submitted_at":"2024-02-01T19:12:25Z","title":"When Benchmarks are Targets: Revealing the Sensitivity of Large Language Model Leaderboards","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01781","snapshot_observed_at":"2026-08-11T15:55:23.051934Z","title":"When benchmarks are targets: Revealing the sensitivity of large language model leaderboards, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10558","last_updated":"2024-12-13T21:03:10Z","snapshot_observed_at":"2026-08-17T23:33:40.123461Z","submitted_at":"2024-12-13T21:03:10Z","title":"Too Big to Fool: Resisting Deception in Language Models","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-11T15:55:23.051934Z"},"links":{"cited_paper":"/paper/2402.01781","citing_paper":"/paper/2412.10558"},"observation_digest":"sha256:eb22d9e7e985bcbf7a7bfd1b1f90715903dbca0b0986ea84f4adaffea6e7c7f0","observation_id":"7fd1be3c-d545-4ff7-98b1-2696f4ee6316","resolution":{"observed_at":"2026-08-11T15:55:23.051934Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01781","last_updated":"2024-07-03T11:20:43Z","snapshot_observed_at":"2026-08-18T18:45:01.785556Z","submitted_at":"2024-02-01T19:12:25Z","title":"When Benchmarks are Targets: Revealing the Sensitivity of Large Language Model Leaderboards","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01781","snapshot_observed_at":"2026-08-11T06:01:13.224508Z","title":"A.; Alnumay, Y.; Alrashed, S.; Alsubaie, S.; Almushaykeh, Y.; Mirza, F.; Alotaibi, N.; Altwairesh, N.; Alowisheq, A.; et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.17874","last_updated":"2025-02-09T16:39:50Z","snapshot_observed_at":"2026-08-16T23:58:19.700639Z","submitted_at":"2024-12-22T09:10:34Z","title":"Evaluating LLM Reasoning in the Operations Research Domain with ORQA","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-11T06:01:13.224508Z"},"links":{"cited_paper":"/paper/2402.01781","citing_paper":"/paper/2412.17874"},"observation_digest":"sha256:98d77556c9b2d7c5c97bc6543c5d07a58407b5f2578597004d62f23481d44287","observation_id":"dc6b0e4d-8ae7-4e3b-b0c6-2ed46a0dbf33","resolution":{"observed_at":"2026-08-11T06:01:13.224508Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01781","last_updated":"2024-07-03T11:20:43Z","snapshot_observed_at":"2026-08-18T18:45:01.785556Z","submitted_at":"2024-02-01T19:12:25Z","title":"When Benchmarks are Targets: Revealing the Sensitivity of Large Language Model Leaderboards","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01781","snapshot_observed_at":"2026-08-08T15:22:44.788604Z","title":"A.; Alnumay, Y.; Alrashed, S.; Alsubaie, S.; Almushaykeh, Y.; Mirza, F.; Alotaibi, N.; Altwairesh, N.; Alowisheq, A.; Bari, M","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06470","last_updated":"2025-02-10T13:50:25Z","snapshot_observed_at":"2026-08-18T02:04:40.850005Z","submitted_at":"2025-02-10T13:50:25Z","title":"A Survey of Theory of Mind in Large Language Models: Evaluations, Representations, and Safety Risks","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-08T15:22:44.788604Z"},"links":{"cited_paper":"/paper/2402.01781","citing_paper":"/paper/2502.06470"},"observation_digest":"sha256:8d67c1c6a03ad01f80fe30c9dbe86bda6caaa9f218d92f4a0d91ae08c9b72695","observation_id":"cc8883a8-ff25-4446-a322-8650f92e5d4f","resolution":{"observed_at":"2026-08-08T15:22:44.788604Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01781","last_updated":"2024-07-03T11:20:43Z","snapshot_observed_at":"2026-08-18T18:45:01.785556Z","submitted_at":"2024-02-01T19:12:25Z","title":"When Benchmarks are Targets: Revealing the Sensitivity of Large Language Model Leaderboards","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01781","snapshot_observed_at":"2026-08-08T15:06:54.722646Z","title":"Saiful Bari, and Haidar Khan","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.722646Z"},"links":{"cited_paper":"/paper/2402.01781","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:8f500d60f5bea1224eb3f7bfdf516136636032ec0d723e16ab5922cebe54db36","observation_id":"6c1b2e4c-07fe-4278-963b-01609ebcdb51","resolution":{"observed_at":"2026-08-08T15:06:54.722646Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01781","last_updated":"2024-07-03T11:20:43Z","snapshot_observed_at":"2026-08-18T18:45:01.785556Z","submitted_at":"2024-02-01T19:12:25Z","title":"When Benchmarks are Targets: Revealing the Sensitivity of Large Language Model Leaderboards","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01781","snapshot_observed_at":"2026-08-08T14:46:29.541762Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06666","last_updated":"2025-02-10T16:52:39Z","snapshot_observed_at":"2026-08-13T02:27:49.966430Z","submitted_at":"2025-02-10T16:52:39Z","title":"Automatic Evaluation of Healthcare LLMs Beyond Question-Answering","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-08T14:46:29.541762Z"},"links":{"cited_paper":"/paper/2402.01781","citing_paper":"/paper/2502.06666"},"observation_digest":"sha256:cff627629bc555aed542830f7561c0d18540a375b544bff45c39fc133abc7c5c","observation_id":"6aabf32b-0f38-4f7f-9150-377d5b946014","resolution":{"observed_at":"2026-08-08T14:46:29.541762Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01781","last_updated":"2024-07-03T11:20:43Z","snapshot_observed_at":"2026-08-18T18:45:01.785556Z","submitted_at":"2024-02-01T19:12:25Z","title":"When Benchmarks are Targets: Revealing the Sensitivity of Large Language Model Leaderboards","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01781","snapshot_observed_at":"2026-08-08T16:11:43.090058Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08662","last_updated":"2025-06-02T04:54:00Z","snapshot_observed_at":"2026-08-18T21:37:50.629904Z","submitted_at":"2025-02-10T09:34:15Z","title":"RoToR: Towards More Reliable Responses for Order-Invariant Inputs","version":3},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-08T16:11:43.090058Z"},"links":{"cited_paper":"/paper/2402.01781","citing_paper":"/paper/2502.08662"},"observation_digest":"sha256:b3a7f3c5fa3ee5527d0066288a63b8bbc396ea372a904bcfab3f9d7a5c0c3467","observation_id":"8075e7fb-d2d2-4739-9b6d-eb560a6331db","resolution":{"observed_at":"2026-08-08T16:11:43.090058Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01781","last_updated":"2024-07-03T11:20:43Z","snapshot_observed_at":"2026-08-18T18:45:01.785556Z","submitted_at":"2024-02-01T19:12:25Z","title":"When Benchmarks are Targets: Revealing the Sensitivity of Large Language Model Leaderboards","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01781","snapshot_observed_at":"2026-08-16T12:33:23.929114Z","title":"When benchmarks are targets: Revealing the sensitivity of large language model leaderboards","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.12562","last_updated":"2025-04-17T01:23:50Z","snapshot_observed_at":"2026-08-19T15:39:14.249790Z","submitted_at":"2025-04-17T01:23:50Z","title":"ZeroSumEval: Scaling LLM Evaluation with Inter-Model Competition","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-16T12:33:23.929114Z"},"links":{"cited_paper":"/paper/2402.01781","citing_paper":"/paper/2504.12562"},"observation_digest":"sha256:657eb4459901b82da9575418988eaf2687b2f28d5c5ecbe4b908fdca13cbf6b4","observation_id":"d59d5cb2-0912-422a-a0cf-c2dece4483e3","resolution":{"observed_at":"2026-08-16T12:33:23.929114Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01781","last_updated":"2024-07-03T11:20:43Z","snapshot_observed_at":"2026-08-18T18:45:01.785556Z","submitted_at":"2024-02-01T19:12:25Z","title":"When Benchmarks are Targets: Revealing the Sensitivity of Large Language Model Leaderboards","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01781","snapshot_observed_at":"2026-08-16T11:17:29.300022Z","title":"When benchmarks are targets: Revealing the sensitivity of large language model leaderboards, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18574","last_updated":"2025-06-11T11:06:08Z","snapshot_observed_at":"2026-08-17T08:53:04.431268Z","submitted_at":"2025-04-22T16:15:19Z","title":"Understanding the Skill Gap in Recurrent Language Models: The Role of the Gather-and-Aggregate Mechanism","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-16T11:17:29.300022Z"},"links":{"cited_paper":"/paper/2402.01781","citing_paper":"/paper/2504.18574"},"observation_digest":"sha256:c6a1fb25dd2269d59e0a86342d84bdf5342b8e9695f62e759722b9af4b18bd81","observation_id":"7d2fff4b-062f-4cdd-a428-2e2513130e46","resolution":{"observed_at":"2026-08-16T11:17:29.300022Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01781","last_updated":"2024-07-03T11:20:43Z","snapshot_observed_at":"2026-08-18T18:45:01.785556Z","submitted_at":"2024-02-01T19:12:25Z","title":"When Benchmarks are Targets: Revealing the Sensitivity of Large Language Model Leaderboards","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01781","snapshot_observed_at":"2026-08-16T05:58:22.018300Z","title":"A., Alnumay, Y., Alrashed, S., Alsubaie, S., Almushaykeh, Y., Mirza, F., Alotaibi, N., Altwairesh, N., Alowisheq, A., Bari, M","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.19561","last_updated":"2025-04-28T08:12:30Z","snapshot_observed_at":"2026-08-19T10:02:26.423225Z","submitted_at":"2025-04-28T08:12:30Z","title":"Quantifying Memory Utilization with Effective State-Size","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-16T05:58:22.018300Z"},"links":{"cited_paper":"/paper/2402.01781","citing_paper":"/paper/2504.19561"},"observation_digest":"sha256:a4713cde3ef08e77441d8fd923fa51dee6535a3aa197a7a9f88a9a0c9e60c0e3","observation_id":"3d6a5638-e8f3-4e0e-9d76-0c8b5cbfff94","resolution":{"observed_at":"2026-08-16T05:58:22.018300Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01781","last_updated":"2024-07-03T11:20:43Z","snapshot_observed_at":"2026-08-18T18:45:01.785556Z","submitted_at":"2024-02-01T19:12:25Z","title":"When Benchmarks are Targets: Revealing the Sensitivity of Large Language Model Leaderboards","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01781","snapshot_observed_at":"2026-08-02T02:14:08.103425Z","title":"A Helpful Assistant","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14399","last_updated":"2026-07-15T22:33:53Z","snapshot_observed_at":"2026-08-18T21:56:03.398677Z","submitted_at":"2026-07-15T22:33:53Z","title":"Instrument Effects in Language-Model Honesty Evaluation: An Auditable Single-System Demonstration","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-02T02:14:08.103425Z"},"links":{"cited_paper":"/paper/2402.01781","citing_paper":"/paper/2607.14399"},"observation_digest":"sha256:b34ecd4dfe28c7d4f547520e4acb57a90720dceb939e99f862df7a5135ba095f","observation_id":"67b6341e-63f4-46a0-8f7d-e48e562108f8","resolution":{"observed_at":"2026-08-02T02:14:08.103425Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2402.01781/citation-record","integrity":"/paper/2402.01781/integrity","json":"/paper/2402.01781/citation-record.json","paper":"/paper/2402.01781"},"outbound":[],"paper":{"arxiv_id":"2402.01781","last_updated":"2024-07-03T11:20:43Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-18T18:45:01.785556Z","submitted_at":"2024-02-01T19:12:25Z","title":"When Benchmarks are Targets: Revealing the Sensitivity of Large Language Model Leaderboards"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"thesis":"As of 22 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 14 inbound Pith citation observations for arXiv:2402.01781."}