{"as_of":"2026-08-09T18:27:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d95bd95616e89acc6d2bc13c573ad5b0c968146ff096ae9b603c8f18c8635a3e","coverage":[{"denominator":107,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T22:54:02.909618Z","state":"measured"},{"denominator":111,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":111,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":11,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":11,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T05:34:26.874878Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T05:57:41.567131Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"cited_work":{"arxiv_id":"2502.04313","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.04313","snapshot_observed_at":"2026-07-03T05:57:41.567131Z","title":"A., Chandra, K","venue":null,"work_id":"3d257c3b-7d9d-4f19-beb3-71558b8b8f39","year":2025},"citing_paper":{"arxiv_id":"2502.05075","last_updated":"2026-04-19T21:23:04Z","snapshot_observed_at":"2026-08-03T04:05:35.638196Z","submitted_at":"2025-02-07T16:46:43Z","title":"Discrepancies are Virtue: Weak-to-Strong Generalization through Lens of Intrinsic Dimension","version":6},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-23T03:24:07.851782Z"},"links":{"cited_paper":"/paper/2502.04313","citing_paper":"/paper/2502.05075"},"observation_digest":"sha256:f06909c5892c6cdc1a11b191943008dc18e173365a1ccf7be0cda8056779959c","observation_id":"2f377797-d54a-44d6-9f25-2988ae19ae4c","resolution":{"observed_at":"2026-05-23T03:25:20.625986Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.04313","snapshot_observed_at":"2026-08-07T05:34:26.874878Z","title":"Chandra, Ponnurangam Kumaraguru, Douwe Kiela, Ameya Prabhu, Matthias Bethge, and Jonas Geiping","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-09T03:05:20.348569Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.874878Z"},"links":{"cited_paper":"/paper/2502.04313","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:c292aec7582ff71f37ba0055915413288b829244d056f9acd358d0046be053ad","observation_id":"f75e2726-15af-4092-bb23-e4c5e4ba78b6","resolution":{"observed_at":"2026-08-07T05:34:26.874878Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.04313","snapshot_observed_at":"2026-08-07T05:27:56.244885Z","title":"A., Chandra, K","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.07962","last_updated":"2025-06-09T17:37:18Z","snapshot_observed_at":"2026-08-08T09:46:08.699885Z","submitted_at":"2025-06-09T17:37:18Z","title":"Correlated Errors in Large Language Models","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-07T05:27:56.244885Z"},"links":{"cited_paper":"/paper/2502.04313","citing_paper":"/paper/2506.07962"},"observation_digest":"sha256:bc86a0193e62610a4b95e8168bd8bf725d4166562b71ece01188a70db4b0b626","observation_id":"7035dbe6-4d82-4cdc-adf5-bbb23b52220d","resolution":{"observed_at":"2026-08-07T05:27:56.244885Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"cited_work":{"arxiv_id":"2502.04313","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.04313","snapshot_observed_at":"2026-07-03T05:57:41.567131Z","title":"A., Chandra, K","venue":null,"work_id":"3d257c3b-7d9d-4f19-beb3-71558b8b8f39","year":2025},"citing_paper":{"arxiv_id":"2604.11609","last_updated":"2026-05-04T02:17:31Z","snapshot_observed_at":"2026-08-09T16:23:10.036912Z","submitted_at":"2026-04-13T15:14:33Z","title":"Intersectional Sycophancy: How Perceived User Demographics Shape False Validation in Large Language Models","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T15:47:31.492842Z"},"links":{"cited_paper":"/paper/2502.04313","citing_paper":"/paper/2604.11609"},"observation_digest":"sha256:8eb0394a19cb5b67caaac048a9e1d471a9093f019f8737a6cdb6ae9eaf9ec3b9","observation_id":"60a4a7cd-7bd7-49e7-a7e4-ff58b5ab653d","resolution":{"observed_at":"2026-05-11T09:50:59.795612Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"cited_work":{"arxiv_id":"2502.04313","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.04313","snapshot_observed_at":"2026-07-03T05:57:41.567131Z","title":"A., Chandra, K","venue":null,"work_id":"3d257c3b-7d9d-4f19-beb3-71558b8b8f39","year":2025},"citing_paper":{"arxiv_id":"2606.11470","last_updated":"2026-06-09T21:59:37Z","snapshot_observed_at":"2026-07-06T23:50:35.052764Z","submitted_at":"2026-06-09T21:59:37Z","title":"The Periodic Table of LLM Reasoning: A Structured Survey of Reasoning Paradigms, Methods, and Failure Modes","version":1},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-06-27T12:59:51.091008Z"},"links":{"cited_paper":"/paper/2502.04313","citing_paper":"/paper/2606.11470"},"observation_digest":"sha256:934af93e5d51f48c7850a535f940b8034683497e651fda6b43c9f86b42fba432","observation_id":"457773f8-fc99-4712-9967-25e6931224a1","resolution":{"observed_at":"2026-07-03T05:57:41.568634Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"cited_work":{"arxiv_id":"2502.04313","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.04313","snapshot_observed_at":"2026-07-03T05:57:41.567131Z","title":"A., Chandra, K","venue":null,"work_id":"3d257c3b-7d9d-4f19-beb3-71558b8b8f39","year":2025},"citing_paper":{"arxiv_id":"2606.28661","last_updated":"2026-06-27T00:37:33Z","snapshot_observed_at":"2026-08-09T14:28:59.457613Z","submitted_at":"2026-06-27T00:37:33Z","title":"When More Sampling Hurts: The Modal Ceiling and Correlation Ceiling of Test-Time Scaling","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-30T09:44:27.786630Z"},"links":{"cited_paper":"/paper/2502.04313","citing_paper":"/paper/2606.28661"},"observation_digest":"sha256:e447996695bd78d12a1137288377a9dee37637fbd833e519eec285e744517c3b","observation_id":"521028b7-62a6-42ed-a508-2e6205bcf38b","resolution":{"observed_at":"2026-06-30T09:44:37.120104Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.04313","snapshot_observed_at":"2026-08-01T15:26:11.571396Z","title":"Great models think alike and this undermines ai oversight","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.18467","last_updated":"2026-07-20T19:37:35Z","snapshot_observed_at":"2026-08-04T10:31:13.766359Z","submitted_at":"2026-07-20T19:37:35Z","title":"Weak-to-Strong Learning in Decision Making","version":1},"reference_index":111,"source":"arxiv_source","source_observed_at":"2026-08-01T15:26:11.571396Z"},"links":{"cited_paper":"/paper/2502.04313","citing_paper":"/paper/2607.18467"},"observation_digest":"sha256:5ce4d46090358182343599605e2637c1b03c7aade037fbd77bc5241398173b45","observation_id":"5efa024b-040e-406c-b4e5-f7d0020ca218","resolution":{"observed_at":"2026-08-01T15:26:11.571396Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.04313","snapshot_observed_at":"2026-08-01T09:29:49.269954Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.20768","last_updated":"2026-07-22T22:34:57Z","snapshot_observed_at":"2026-08-07T08:12:23.947409Z","submitted_at":"2026-07-22T22:34:57Z","title":"Are Diversity Metrics Measuring Diversity? A Capability-Controlled Audit of Majority-Vote Gain in LLM Ensembles","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-01T09:29:49.269954Z"},"links":{"cited_paper":"/paper/2502.04313","citing_paper":"/paper/2607.20768"},"observation_digest":"sha256:e4aa1f6e1214673bf5219e58b7dd6880d5566c5fa8708a90cff321feba6fa0f2","observation_id":"73078668-45a8-44b3-aa29-74ed547193f8","resolution":{"observed_at":"2026-08-01T09:29:49.269954Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.04313","snapshot_observed_at":"2026-08-01T03:57:40.784183Z","title":"TestForge: Feedback-Driven, Agentic Test Suite Gen- eration","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.23002","last_updated":"2026-07-25T02:38:07Z","snapshot_observed_at":"2026-08-09T04:08:38.159012Z","submitted_at":"2026-07-25T02:38:07Z","title":"Adversarial Test-Hardening for AI-Written Code: An Instrument Autopsy and a Pre-Registered Causal Estimate of the Critic Loop","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-01T03:57:40.784183Z"},"links":{"cited_paper":"/paper/2502.04313","citing_paper":"/paper/2607.23002"},"observation_digest":"sha256:090ce3610d5b48d5323823c699f60e5c6a1298088cb01833a08586a6a49a283e","observation_id":"5edd23c8-e32f-443d-abcc-7c43b8f7eec3","resolution":{"observed_at":"2026-08-01T03:57:40.784183Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.04313","snapshot_observed_at":"2026-08-03T10:20:13.027463Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.29274","last_updated":"2026-07-31T10:44:10Z","snapshot_observed_at":"2026-08-08T07:06:38.636808Z","submitted_at":"2026-07-31T10:44:10Z","title":"Language Models Agree With Each Other, Not With Readers","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-03T10:20:13.027463Z"},"links":{"cited_paper":"/paper/2502.04313","citing_paper":"/paper/2607.29274"},"observation_digest":"sha256:c5e282d01169b03e148fd95ea8f162d47e4ff308f4d8652221ac2dbb68674304","observation_id":"9664da98-6e47-449e-ac37-88e30918f897","resolution":{"observed_at":"2026-08-03T10:20:13.027463Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.04313","snapshot_observed_at":"2026-08-04T22:38:10.185108Z","title":"Goel et al","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.01704","last_updated":"2026-08-03T05:09:49Z","snapshot_observed_at":"2026-08-08T02:28:04.716439Z","submitted_at":"2026-08-03T05:09:49Z","title":"Floor, Ceiling, and the Fusion Gap: How Much of Crowd Reading Attention Can Machines Predict?","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-04T22:38:10.185108Z"},"links":{"cited_paper":"/paper/2502.04313","citing_paper":"/paper/2608.01704"},"observation_digest":"sha256:5357be08e70b741c465f2e41449adff37475711092f3ceb789616b5552df1c98","observation_id":"41bdb891-3502-4a91-92ec-d5d299d47bce","resolution":{"observed_at":"2026-08-04T22:38:10.185108Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2502.04313/citation-record","integrity":"/paper/2502.04313/integrity","json":"/paper/2502.04313/citation-record.json","paper":"/paper/2502.04313"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.499159Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.499159Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:b4f35df8836c09e663c44830155c8ab51583033060c8572293e3668ea6b3acec","observation_id":"351a9710-201a-4bdf-90ae-22f274d07cd7","resolution":{"observed_at":"2026-08-08T22:54:02.499159Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.505415Z","title":"B., Lozhkov, A., Bakouch, E., Blázquez, G","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.505415Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:eed03cdf546fe50854bb6d424e4b147d80b5087568fadcda6cd9314335419869","observation_id":"bc38d202-aded-49e1-9ff5-cd10b94b19ab","resolution":{"observed_at":"2026-08-08T22:54:02.505415Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.510124Z","title":"and Perez-Villadoniga, M","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.510124Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:d03edbf005bd0ff01da3b41c69b110a71d43540e1dc55f6c01c5b363c0e84976","observation_id":"10a20e74-d1d6-4b55-b33b-7565d8dbb01a","resolution":{"observed_at":"2026-08-08T22:54:02.510124Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.514664Z","title":"Towards evaluations-based safety cases for ai scheming, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.514664Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:61a037683bcf85322b9ba41b0f24676ef7a9736744dd277a98cc0b5e8ab12430","observation_id":"89062b19-45b1-4226-a52e-0cba059ca941","resolution":{"observed_at":"2026-08-08T22:54:02.514664Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.518935Z","title":"Revisiting model stitching to compare neural representations","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.518935Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:99d7b7b201666c3a5298a246d9d8bbea172aa8960496560600aef5c131ece852","observation_id":"fe68c7e8-f373-421d-840d-4987804141cd","resolution":{"observed_at":"2026-08-08T22:54:02.518935Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.523516Z","title":"F., Ammanamanchi, P","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.523516Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:fdd9649ca8cd2a962202705b2db0b079abe0f70c5d41bae006ae2269dcd555ab","observation_id":"7344e4e3-9b00-4626-90e6-7c2123e8eda2","resolution":{"observed_at":"2026-08-08T22:54:02.523516Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.527965Z","title":"Holistic evaluation of language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.527965Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:dd366da97f2a650e77b86efa54b8f0988717dd805385d7613a2e96a38a5bb6f5","observation_id":"624b9ca3-c429-4534-88a9-828f4b9d4514","resolution":{"observed_at":"2026-08-08T22:54:02.527965Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.532560Z","title":"Which prompts make the difference? data prioritization for efficient human llm evaluation, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.532560Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:34478fd54093149c8731304f248834cea910797cfcad17ac39b3e5f5c77067c9","observation_id":"d6067d91-2756-42a3-87e7-fa7e08141be2","resolution":{"observed_at":"2026-08-08T22:54:02.532560Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.536524Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.536524Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:83f73af404f2b5e55d3f7071d3e610eb5abd2f52e558f68928d131cfe8de4b82","observation_id":"7ca69eac-b5cd-41c4-893b-f0049538c184","resolution":{"observed_at":"2026-08-08T22:54:02.536524Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.540744Z","title":"D., Martinez-Plumed, F., Tenenbaum, J","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.540744Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:0f16137e67b2f6acf3a986ee4252892b4fd1b8c18f60182a6679fcedb74de631","observation_id":"f29e235b-10ba-4be1-9c6f-41f213132658","resolution":{"observed_at":"2026-08-08T22:54:02.540744Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.544921Z","title":"H., Baker, B., Gao, L., Aschenbrenner, L., Chen, Y., Ecoffet, A., Joglekar, M., Leike, J., Sutskever, I., and Wu, J","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.544921Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:36c887760dde8feacbdf9fce911d92dd4137b0e5bc635d0819a0f4de5ba99fca","observation_id":"e5b1e0a9-87f3-4ad6-8b5e-275672afead2","resolution":{"observed_at":"2026-08-08T22:54:02.544921Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.549028Z","title":"A portfolio approach to research funding","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.549028Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:627de7ffa6c6b3449a9c0279eaba904dcf7c4293311670446e54e274a9396da5","observation_id":"5463a38a-151a-42ab-911c-99f1b01f1435","resolution":{"observed_at":"2026-08-08T22:54:02.549028Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.553066Z","title":"Quantifying the gain in weak-to-strong generalization","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.553066Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:3b3a5f6ab3aed13f3de0beb3b33b2fc2938d204e3d7fe6e3272704ee9a63cd74","observation_id":"32d9e6ff-23c4-407f-8081-3afa6e918316","resolution":{"observed_at":"2026-08-08T22:54:02.553066Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.557083Z","title":"H., Chen, S., Liu, Z., Jiang, F., and Wang, B","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.557083Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:60c83ac93a407ab3dcc2c8f1a30ef6297429bec95454e57b4099786d53913d81","observation_id":"084cabde-0eff-45f9-aeaa-1edf959799c1","resolution":{"observed_at":"2026-08-08T22:54:02.557083Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.561178Z","title":"E., Stoica, I., and Xing, E","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.561178Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:7e0761edefe77f2b748d086874b3df6fd85a3c8ac1dcc77bf7ef9e41e0c40bfd","observation_id":"2a37be34-8b04-4434-b4e8-288471cf7a27","resolution":{"observed_at":"2026-08-08T22:54:02.561178Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.565311Z","title":"J., and Jurman, G","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.565311Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:d9766c85c633872e5475a6ae05131cfa93402f97bfd9de828f6126d7879b5e6e","observation_id":"24d6c2de-4546-48b9-aef6-256df238b6e4","resolution":{"observed_at":"2026-08-08T22:54:02.565311Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.569270Z","title":"B ool Q : Exploring the surprising difficulty of natural yes/no questions","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.569270Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:ea20ddff9092c98b0005295a921585a23ce90cf2c9d7c77bcfb5da175635e776","observation_id":"ecd41ed7-e481-4083-9296-187d7425cec7","resolution":{"observed_at":"2026-08-08T22:54:02.569270Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.573510Z","title":"A coefficient of agreement for nominal scales","venue":null,"work_id":null,"year":1960},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.573510Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:56e7d50baad5153c882629ede01759daf8cd9ca717a7ea1a9b3fc1f9e6b349da","observation_id":"28e6491f-b70e-4aad-a32a-8e60fa5b7d8b","resolution":{"observed_at":"2026-08-08T22:54:02.573510Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.577701Z","title":"X., Taori, R., Zhang, T., Gulrajani, I., Ba, J., Guestrin, C., Liang, P","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.577701Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:e373a4fcad459edc19732f74016f223d2a848cbe9e4eb2ca4991a98fe09f1eab","observation_id":"d5b12e98-b4f1-4e91-b267-980bbb025e79","resolution":{"observed_at":"2026-08-08T22:54:02.577701Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.581768Z","title":"Length-controlled alpacaeval: A simple debiasing of automatic evaluators","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.581768Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:379632cb06700f0379aa5bfc409f646b93a60c27a2e73801269f97520c6d72b2","observation_id":"1101ec06-8c75-45ee-bf4a-28395cdcc416","resolution":{"observed_at":"2026-08-08T22:54:02.581768Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.585709Z","title":"E., and Yeung-Levy, S","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.585709Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:b56ce089d087f337f54f8d3f977d8b3f9504d8349ca9408084964199381961e6","observation_id":"75528b9c-8fd8-474f-9c6e-ae9f344c4aaf","resolution":{"observed_at":"2026-08-08T22:54:02.585709Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.589693Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.589693Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:ddcd922283da1d3b55faa0603931cbdda0a6bac76a093c3bfb7937e6889ee6ba","observation_id":"a66b2253-dc09-4891-91e9-1ced40e36761","resolution":{"observed_at":"2026-08-08T22:54:02.589693Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.593501Z","title":"Accuracy is not all you need","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.593501Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:25fc0182c452e1ae1ee0f39dfd10029fe6e26b4f4c6aa5676f83994076e7f1cd","observation_id":"4afd2635-189d-4323-b26e-97f240d96513","resolution":{"observed_at":"2026-08-08T22:54:02.593501Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.597561Z","title":"Model changelists: Characterizing updates to ml models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.597561Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:90c7546a3565f997036ddb68bbfa02e2bd88966f3f91c323eea595878d8c4a3e","observation_id":"a35b7e18-5bfd-4714-ba92-8451c889f33e","resolution":{"observed_at":"2026-08-08T22:54:02.597561Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.601484Z","title":"L., Levin, B., Paik, M","venue":null,"work_id":null,"year":1981},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.601484Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:dcf981d6beab23bda169198caa9f5ffb2d25e4794c4c77d3984e3e4725e502f9","observation_id":"5d3ca6ba-958d-43fd-b3fe-1a82d3238d80","resolution":{"observed_at":"2026-08-08T22:54:02.601484Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.605595Z","title":"Evaluating superhuman models with consistency checks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.605595Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:f1873d07abab2dc35ab4dd8aa194e11fd01d0f9706f56427baff29afe6acd1a6","observation_id":"fdf8696a-8292-459a-b836-9c3b6403c427","resolution":{"observed_at":"2026-08-08T22:54:02.605595Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.609629Z","title":"A framework for few-shot language model evaluation, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.609629Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:270bd5ce875f8cae6cfa0417077ad379f879ea0acb556e2fa1a639dc2d7d2440","observation_id":"88e5db68-1477-4769-a059-f159ed67112d","resolution":{"observed_at":"2026-08-08T22:54:02.609629Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.613633Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.613633Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:5f634b9528b27e327b1b10f5d2ce2750afed177b24ceec09e870502e7a267db3","observation_id":"22b7d1ca-91a2-49cc-80c8-1677fbf80df5","resolution":{"observed_at":"2026-08-08T22:54:02.613633Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:04.213935Z","title":"A., and Brendel, W","venue":null,"work_id":"f1d50f9a-e24b-4755-86d8-ea865fb2bbb7","year":2021},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.617652Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:37d64910caf892a365fa16d19f653bc969e63da7a2f49a5fc21ec253d6a6bae2","observation_id":"56bb0f04-c84f-4d33-9564-4ebbf35f5235","resolution":{"observed_at":"2026-08-08T22:54:04.218211Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:04.200317Z","title":"Gemma 2: Improving open language models at a practical size, 2024","venue":null,"work_id":"9619f908-37d8-4137-89ed-7a85e102baca","year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.621699Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:a6dea0ce1062e140050251ef6304fb2b3df12025ce2e38de8dbbe0c7aa50fae2","observation_id":"2fa2f3f2-dc7a-49eb-849f-2439bdb693aa","resolution":{"observed_at":"2026-08-08T22:54:04.204866Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:04.186318Z","title":"Onebench to test them all: Sample-level benchmarking over open-ended capabilities, 2024","venue":null,"work_id":"a27af99b-ef11-431e-b676-2ed576a05812","year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.625821Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:a4564dcfe1f2118fb115d8c6ce9cd375b8cde1f89d0b7da1d5ef03788821dd97","observation_id":"91b98932-cf8c-4aad-89b9-5ba30303ba90","resolution":{"observed_at":"2026-08-08T22:54:04.191341Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:04.172760Z","title":"Chatgpt outperforms crowd workers for text-annotation tasks","venue":null,"work_id":"739878af-ab25-46c9-82a3-76aae9a1fe88","year":2023},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.629816Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:2f8f967eb19ba35cd4c083e3431920478f7255dce4c8fde336d94239f317f3a5","observation_id":"1f06f59e-2779-469f-a538-15c138c5c6d6","resolution":{"observed_at":"2026-08-08T22:54:04.177199Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:04.159215Z","title":"and Dao, J","venue":null,"work_id":"93079c97-ea47-459b-bcc4-07a04bfa5a23","year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.633806Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:0eb759db70289bfa7cc71be99a7d4c42d09ef5e9313e9401bf5e1ec3af0b38e0","observation_id":"fc14ef14-2eaa-4b61-a0ab-a8ed12c72de8","resolution":{"observed_at":"2026-08-08T22:54:04.163478Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.637905Z","title":"and Dao, T","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.637905Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:6a0f87f4523aac854c1b8d51fd97e454585430b006b30e71e54370ace19a439f","observation_id":"a427666a-3b47-4300-b504-162d2bf457e2","resolution":{"observed_at":"2026-08-08T22:54:02.637905Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:04.136513Z","title":"Vision superalignment: Weak-to-strong generalization for vision foundation models, 2024","venue":null,"work_id":"85dec5aa-fed2-479b-9b45-c0e82e6a992e","year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.641949Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:c328faf9b8c508b267006a66d16cf00ef01ceff4c819633e40fc1d5cf9719cc7","observation_id":"1523b8ad-7dde-4f9c-95a1-260b23b97278","resolution":{"observed_at":"2026-08-08T22:54:04.140966Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:04.123392Z","title":"On the blind spots of model-based evaluation metrics for text generation","venue":null,"work_id":"b660b12e-4413-4579-8178-c30add3092e9","year":2023},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.645865Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:c3202474161e373ad758f4c394217d83bfb1aa5868570aa750507623971277dc","observation_id":"577b46d6-cd14-491c-939b-27ca786e87a2","resolution":{"observed_at":"2026-08-08T22:54:04.127725Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:04.110127Z","title":"Aligning AI with shared human values","venue":null,"work_id":"93241234-5f13-4ecf-9fcc-91796a7c093e","year":2021},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.649760Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:af0aff22bf95a816f160392f39cddcd564150dcf2424ab5411f85d092a95e71b","observation_id":"aaca59ab-f8ae-4b2e-ac1e-4c3e91210d95","resolution":{"observed_at":"2026-08-08T22:54:04.114371Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:04.095996Z","title":"Measuring massive multitask language understanding","venue":null,"work_id":"b6803d4d-ffd0-4edf-8a9c-315b5fcebca3","year":2021},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.653975Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:88bc73598ff98aba4d6a58a4070ae8f64bd1d3888c4099635a63d13b03ead8ff","observation_id":"011a5697-ca30-46e0-8686-78f9ab291e95","resolution":{"observed_at":"2026-08-08T22:54:04.100840Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:04.081765Z","title":"J., Shen, Y., Wallis, P., Allen - Zhu, Z., Li, Y., Wang, S., Wang, L., and Chen, W","venue":null,"work_id":"202e4de7-5505-44ef-9975-9b934cd3a6d6","year":2022},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.657997Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:5f59719ffad3317a39a0b593b7abed48f2780be1c84be2be8ad0abc3af3546f0","observation_id":"e64800b2-7fa1-459f-aaae-e297f456c768","resolution":{"observed_at":"2026-08-08T22:54:04.086162Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:04.067748Z","title":"Cosmos QA : Machine reading comprehension with contextual commonsense reasoning","venue":null,"work_id":"45a9af5f-b3a0-47a5-a03d-a0d1ed1d1f9c","year":2019},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.662004Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:bb1302193b26ab2fd761060ed8892e7ebfafd08c339bce0ca95f4633d14629a0","observation_id":"9ddc60a5-eb3b-4892-8166-edfe57ce0369","resolution":{"observed_at":"2026-08-08T22:54:04.072418Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:04.054405Z","title":"D., Parker-Holder, J., Behbahani, F., Mavalankar, A., Shi, Y., Schaul, T., and Rockt\\\" a schel, T","venue":null,"work_id":"3b807fe3-4e6e-463e-9c6b-cb5c6541b93b","year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.666026Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:ac8a344513586b6748cfb7541a1c5b5396fb4b31e5af409fcb66f42465ff16bb","observation_id":"f6919b46-2428-4dab-aadb-e1177b86fac4","resolution":{"observed_at":"2026-08-08T22:54:04.058759Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:04.040456Z","title":"Position: the platonic representation hypothesis","venue":null,"work_id":"173d10a6-780f-474f-b59f-40356d856ac5","year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.670428Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:f58b6a8d883f79cc160390c7097d90ea45e4cdaf5cd3c5b83fc7398a82b058ca","observation_id":"14bca023-6aee-4e68-9801-66c0a6bcfb53","resolution":{"observed_at":"2026-08-08T22:54:04.045168Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:04.027118Z","title":"X., Wexler, J., Reif, E., Kallarackal, K., Chang, M., Terry, M., and Dixon, L","venue":null,"work_id":"69e7f1d3-8c20-4f38-a8df-1df1a8788c91","year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.674409Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:32ea87b9f2cdddc52051e4a3e2983e199fa986fc3fa6a873b9d2c1f8c96371bc","observation_id":"5396c51a-4ce2-4242-858e-762990d8d36c","resolution":{"observed_at":"2026-08-08T22:54:04.031544Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.678656Z","title":"B., Chess, B., Child, R., Gray, S., Radford, A., Wu, J., and Amodei, D","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.678656Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:006057bfdd8bf9087bcf74c76fc57bffc00e07a0430cb56d4ee823e25820deb4","observation_id":"2aca9ea4-44af-4917-afe9-d1cf0782afd5","resolution":{"observed_at":"2026-08-08T22:54:02.678656Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:04.005139Z","title":"L., and Koyejo, S","venue":null,"work_id":"7b2c2e25-eb80-4343-a155-9844d4ef9de2","year":2025},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.682664Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:bf690f608df2d9811eeb05f5b7e8a4eed787221dcc0c0e7df9812437f5871f5c","observation_id":"acc281c0-7206-4cb5-8003-b24bf1fa8bf6","resolution":{"observed_at":"2026-08-08T22:54:04.009467Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.991615Z","title":"Y., Kram\\' a r, J., Brown-Cohen, J., Albanie, S., Bulian, J., Agarwal, R., Lindner, D., Tang, Y., Goodman, N","venue":null,"work_id":"1cc045fb-de67-4c4e-a93e-f085237d42f9","year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.686682Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:64e215a59f62706e297c0a96eba0daf3cbdf7bf57d8b9a11af16c69a8031e79f","observation_id":"bc67f1dd-d232-4ef3-b008-d546ed07c630","resolution":{"observed_at":"2026-08-08T22:54:03.996002Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.978241Z","title":"Looking beyond the surface: A challenge set for reading comprehension over multiple sentences","venue":null,"work_id":"554d867d-6b40-41d4-80f7-1ecd02d24899","year":2018},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.690615Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:f40eee6f8e457d6ae1e418cac39073687bac891373c0dc7ad1a6fdf2b4bbd21b","observation_id":"12504cf5-ffd8-44a4-b842-26f512d89830","resolution":{"observed_at":"2026-08-08T22:54:03.982653Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.964764Z","title":"Similarity of neural network models: A survey of functional and representational measures","venue":null,"work_id":"06a8822b-ec75-4cf5-bbb1-fbd6282f979b","year":2025},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.694654Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:32aea398720b7f846536467ec7cb69916962c9b9f37755fe09dc9362e91000e9","observation_id":"d274a51d-3517-4daf-803b-99267822a003","resolution":{"observed_at":"2026-08-08T22:54:03.969140Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.698830Z","title":"and Raghavan, M","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.698830Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:88e941b7132fc0fde6c0001cdfad5eb6201cdf341605ba170766cfd5c3a008c2","observation_id":"99200a00-7cf0-454f-bc8b-e4f7501e1275","resolution":{"observed_at":"2026-08-08T22:54:02.698830Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.941712Z","title":"To ship or not to ship: An extensive evaluation of automatic metrics for machine translation","venue":null,"work_id":"23bd6b59-5298-49a5-8acd-b02b0129e9a5","year":2021},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.703251Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:1e8a8208df3ada06dd0e5fe0b8bcd976ba849f2087a254416449b915fe5ae493","observation_id":"cec46f14-7674-4a28-935a-2ca32d7d5ad5","resolution":{"observed_at":"2026-08-08T22:54:03.946538Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.928333Z","title":"R., Vaidya, A., Mahmood, F., Zitnik, M., Chen, T., and Hartvigsen, T","venue":null,"work_id":"3b7267a6-1063-4f20-84e8-d845069b4aad","year":2025},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.707242Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:55d630dedaadbf361bd8c97f0b05854417f993711e97df09c47ddeb81a03a9ed","observation_id":"05fb5b76-13a7-444a-9df8-107308c4f602","resolution":{"observed_at":"2026-08-08T22:54:03.932813Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.914792Z","title":"I., Kim, Z","venue":null,"work_id":"f26f8b95-fa0d-405f-919f-2e963280b27d","year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.711299Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:068371fd0f17e43719d7f1aa9d9bcdc7fa03944bebca2308131d5217c8f5e3ed","observation_id":"e009bac9-3949-4fc9-9c3d-f15ad50d6fa7","resolution":{"observed_at":"2026-08-08T22:54:03.919152Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.901003Z","title":"Similarity of neural network representations revisited","venue":null,"work_id":"d3228247-f48c-4954-b536-62d3c700a57b","year":2019},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.715539Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:3f4274676f365e80d17cfe79cb465bf0e05b142ce6bab732bf668f33f8a4dbed","observation_id":"aae7df1f-99f2-4617-b531-964b0cf21a23","resolution":{"observed_at":"2026-08-08T22:54:03.905440Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.886969Z","title":"Reliability in content analysis: Some common misconceptions and recommendations","venue":null,"work_id":"2953e364-8c66-494f-8310-e3a7552b9dcb","year":2004},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.719705Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:eb6d42807f0b400bd3a85aacc0f772fb58e55941734d94cae93f432f998eb800","observation_id":"391c8edd-cc8f-494c-8082-bc183cfc1955","resolution":{"observed_at":"2026-08-08T22:54:03.891595Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.723706Z","title":"H., Gonzalez, J","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.723706Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:b60db950275c208f61fceae0a213fc8ff56cc21f590afcb029a997bedcaff41f","observation_id":"6669ddae-4019-4e0c-9618-21c3daf19486","resolution":{"observed_at":"2026-08-08T22:54:02.723706Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.865036Z","title":"D., Dombrowski, A.-K., Goel, S., Mukobi, G., Helm-Burger, N., Lababidi, R., Justen, L., Liu, A","venue":null,"work_id":"b0ae3126-94db-4e35-a5d8-0851f0e837bc","year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.727875Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:54e0761c98737a1af16ede2dfbbd85e652ecf4795ad87668a9e7e9db20f36f64","observation_id":"aff27eea-b36d-49a2-9410-1c9eac791ca2","resolution":{"observed_at":"2026-08-08T22:54:03.869351Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.852368Z","title":"E., and Stoica, I","venue":null,"work_id":"f897eff6-1232-4eee-bf2c-848f5a1a2cd7","year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.731876Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:4d67f0e300e0f4ea8c800368628bda7582e3d23abecff18de6bc05fa99ad8a82","observation_id":"ef499afc-6151-4c0c-af17-1976f19a17a1","resolution":{"observed_at":"2026-08-08T22:54:03.856488Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.839472Z","title":"D., Gunasekar, S., and Lee, Y","venue":null,"work_id":"669996ef-5448-40fa-9f4e-255c6a88a24a","year":2023},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.736143Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:04ce6031224106ebe743490185fa83632224037b173340e3a71f06b16fbf044e","observation_id":"da7f1fe0-081d-4e5e-a00c-9788ef24291b","resolution":{"observed_at":"2026-08-08T22:54:03.843754Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.826753Z","title":"Let's verify step by step","venue":null,"work_id":"970d5201-b6a0-4d85-a1f4-c5b75bae4438","year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.740279Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:ef0142499975b06952c219acb8b714500f5dbd8062a5295835f53d6268243b45","observation_id":"0087be4f-4050-4954-ab04-ef6a462875cc","resolution":{"observed_at":"2026-08-08T22:54:03.830938Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.814103Z","title":"LLM s as narcissistic evaluators: When ego inflates evaluation scores","venue":null,"work_id":"c8ee408b-96e0-44b5-b3d6-566c907bc710","year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.744368Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:fb38547482385d62512bbd02ce479ddd1efb8a7991805f5621ba293b17c62fca","observation_id":"3fee68f5-8e25-4a7d-a80a-9fb13686e5c5","resolution":{"observed_at":"2026-08-08T22:54:03.818229Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.801026Z","title":"The llama 3 herd of models, 2024 a","venue":null,"work_id":"29254e5a-3626-4e06-83c0-6f97b7897c9b","year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.748523Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:2bd5ade166878fe3715e1866225668bebc6160887f34e5da8d59c6f406fa8ade","observation_id":"19826aad-942b-497f-852b-3ab3ced49dff","resolution":{"observed_at":"2026-08-08T22:54:03.805223Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.788249Z","title":"Llama 3.2 model card","venue":null,"work_id":"d576846c-318d-4e67-9a41-58c075e16713","year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.752626Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:44ab6d7f3c4ccaaef42dae1f2ccd166aa62b80b4db7e1baf4505c4c656011fda","observation_id":"f42c26ec-d16a-44df-99c2-f4f7be20b5e4","resolution":{"observed_at":"2026-08-08T22:54:03.792392Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.775512Z","title":"Llama 3.3 model card","venue":null,"work_id":"ab52a23e-4329-4824-b624-0a0fa139a006","year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.756683Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:39e18ed5d72bda75aeffcf39b801b6f5bbe9df98c3d2ad4efbcbe3a174fef016","observation_id":"821877d2-719d-4546-ae55-0a1ac1066388","resolution":{"observed_at":"2026-08-08T22:54:03.779654Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.762175Z","title":"Multi-agent actor-critic for mixed cooperative-competitive environments","venue":null,"work_id":"c34f82f6-8d54-46dd-a550-ccfa56f0eb2e","year":2017},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.760827Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:921a4f30e057513a00400fbf98499d0948c54cb18de43f1ebce6cc4cc103473f","observation_id":"950325e3-7131-4f2e-ad0f-b84082a1026e","resolution":{"observed_at":"2026-08-08T22:54:03.766461Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.748177Z","title":"An adversarial perspective on machine unlearning for AI safety","venue":null,"work_id":"3ff12f4d-849c-4695-9c49-ec25393c9996","year":2025},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.764898Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:92791983cfb03a09cd3f5147eb5c48b6133ad76a071abf8a99ba2591e9ac05c1","observation_id":"37d5a215-6c58-4727-8221-83a378d7ea81","resolution":{"observed_at":"2026-08-08T22:54:03.753002Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.733806Z","title":"Aidanbench: Stress-testing language model creativity on open-ended questions","venue":null,"work_id":"2e3ed829-8ca2-477d-8f30-51fbd238d98a","year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.771222Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:714651e9ad387613a884750193f3f692005b325e8ddf3ad2d7a608ee97d5b9ad","observation_id":"026067bc-61f9-4dcf-b7df-c9d80e3cc79b","resolution":{"observed_at":"2026-08-08T22:54:03.738463Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.720617Z","title":"Phi-4 technical report","venue":null,"work_id":"a184959b-5e4c-4ca1-882e-a94af0cf9764","year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.775437Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:14606e71d2384bfb28f8090af1941ac96ddf390e64bbcf930c0d6ae164004573","observation_id":"36930aaa-6753-4d3b-805a-242a039a054d","resolution":{"observed_at":"2026-08-08T22:54:03.724854Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.707490Z","title":"Ministral 8b instruct model card","venue":null,"work_id":"04396548-bc42-4962-971b-63fe1b6e20c1","year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.779571Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:f869995b2bfeaa7519b186b367d897d6a48ccb754214cf98e3d81ada6b44a935","observation_id":"ae7f2323-a1a7-40d0-a21c-26f49cd5ea7a","resolution":{"observed_at":"2026-08-08T22:54:03.711740Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.694019Z","title":"M., and Shen, Z","venue":null,"work_id":"d8ec09c4-1efb-4816-b4bf-718c02b6f90e","year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.783597Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:6b7a36e0245bbfec4af4bcef5bea7ee10f78762015300e86320846046a30c8a6","observation_id":"56b076bc-a7b1-429d-8f48-c2edc75bf658","resolution":{"observed_at":"2026-08-08T22:54:03.698254Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.680079Z","title":"Adversarial NLI : A new benchmark for natural language understanding","venue":null,"work_id":"5ead1063-d67c-42f3-943c-edc14bb59a97","year":2020},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.787661Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:b4cd6f8c29e12b46890a813ea2708d3573869d521842e2e0aba4bf7ec86d574b","observation_id":"4f2031d4-2663-4d7e-a8de-6f0f128cad89","resolution":{"observed_at":"2026-08-08T22:54:03.684699Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.791538Z","title":"Gpt-4 technical report, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.791538Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:e1359dfbee8a50f60e4dc19ccaf402883ca9cb13d962b1420b7ef3f8aa2ca28e","observation_id":"328b9d0d-6cc0-4bbb-bd07-79496927ccad","resolution":{"observed_at":"2026-08-08T22:54:02.791538Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.656601Z","title":"F., Leike, J., and Lowe, R","venue":null,"work_id":"7e6e4647-a5aa-493a-969a-14bb5f572cc9","year":2022},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.795542Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:587561ef851af247cd39579e64f786e4b79401f208d3bb8b838a0f51ab247665","observation_id":"66d365a5-05ea-4b03-aaaa-4062a962255a","resolution":{"observed_at":"2026-08-08T22:54:03.661452Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.642340Z","title":"R., and Feng, S","venue":null,"work_id":"82e2326a-44a8-4288-943a-9a7f7cdb6068","year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.799619Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:03a7ac092054d29acdc22f6340ac3db76ec46f9717b16b6de20bf4413421ebca","observation_id":"3a369e5c-8f59-4496-8690-319492825dae","resolution":{"observed_at":"2026-08-08T22:54:03.646836Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.628781Z","title":"B leu: a method for automatic evaluation of machine translation","venue":null,"work_id":"65bd0fe1-e616-4a25-8d9e-0a5346f9a5eb","year":2002},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.803793Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:80a2b8575313386dbe1933000286b918ec0ab88d42c29c8a802b2896d50b51ab","observation_id":"2abf2ede-959b-422a-88ff-419eda4c13b8","resolution":{"observed_at":"2026-08-08T22:54:03.633349Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.615021Z","title":null,"venue":null,"work_id":"3dc1a261-cd09-4e29-a822-0582ae09ac8a","year":2019},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.807849Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:26922fc243daae52aeb1e71c186f6842654b6588ab0bd78863cace13cce0fbc7","observation_id":"ee4d54a4-1ea5-4858-b3d6-eea30d8a629e","resolution":{"observed_at":"2026-08-08T22:54:03.619371Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.601152Z","title":"Mauve: Measuring the gap between neural text and human text using divergence frontiers","venue":null,"work_id":"b6370459-1338-4b1d-b63e-0492435dc857","year":2021},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.811963Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:75649ddeff6bb2e7924c5ebd7ab544a773356973101e29a916bf01f57b2c1fb2","observation_id":"ad3824cc-3727-4ee6-89ab-6c116585eff8","resolution":{"observed_at":"2026-08-08T22:54:03.605737Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.587139Z","title":"Qwen2.5 technical report, 2025","venue":null,"work_id":"f073b297-297f-47b3-9bb2-e975b47487bf","year":2025},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.815926Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:99a2dc0d601543f13c1a962ba9b599da2d22fcc2d27c3939fb51bedc64a52fd7","observation_id":"9309217f-4d70-46d3-95ee-d8e5eb3237d7","resolution":{"observed_at":"2026-08-08T22:54:03.591495Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:02.819965Z","title":"Language models are unsupervised multitask learners, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.819965Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:d1d71c88d326e68a5a19ef37d118c45f56585bab06db737c3e06273107ceeb46","observation_id":"cab7d1eb-4878-4b71-8bd1-b0bffb28679e","resolution":{"observed_at":"2026-08-08T22:54:02.819965Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.564773Z","title":"Getting closer to ai complete question answering: A set of prerequisite real tasks","venue":null,"work_id":"f495735e-b0bf-449e-a192-607106399b73","year":2020},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.823883Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:602281855b14e86a39b4d857ef345e157ad17741620eaaed8edefdc6a87c2fdd","observation_id":"1e90a77f-c52c-4d8d-8f21-eb322a4ed88e","resolution":{"observed_at":"2026-08-08T22:54:03.569218Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.551653Z","title":"S., Vinyals, O., H \\' e naff, O","venue":null,"work_id":"4145fcff-9aee-4e07-bb7e-1d2f8a42c34c","year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.827877Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:8b3e1b8b4a00639d8ab68361c13b8b1ebd8d926b5ad51ff58748e2813493bf67","observation_id":"afb1a7e9-d819-4631-975c-b82f00feb3ec","resolution":{"observed_at":"2026-08-08T22:54:03.555784Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.537946Z","title":"Min-mid-max scaling, limits of agreement, and agreement score, 2020","venue":null,"work_id":"193b3c24-324a-4827-85b6-e8fc04125312","year":2020},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":81,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.831879Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:583c93b75fedf4812752f1080c97928da742b7308e5859895a5e2721fc5118a2","observation_id":"d5265bf4-8d01-498f-9ec9-2e1da7b37e89","resolution":{"observed_at":"2026-08-08T22:54:03.542611Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.524243Z","title":"Social IQ a: Commonsense reasoning about social interactions","venue":null,"work_id":"5de41d79-331e-47fe-b272-3b89ca53742f","year":2019},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":82,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.836051Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:05be0f60ded9dbd135770f14023998e810c8344ad8bd4addc07b763defc65134","observation_id":"82105bf6-11b5-4f56-9a45-52dcc379f91f","resolution":{"observed_at":"2026-08-08T22:54:03.529048Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.510489Z","title":"Experiments in weak-to-strong generalization, 2024","venue":null,"work_id":"b63eb175-5b96-4449-9df8-0b4ad3cee2fc","year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.840396Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:6bad47e86dca1c83f18d8015af0fb232735a14c5fc0ae86493fe680dbffa1807","observation_id":"938acea5-090e-442f-9df2-97cfa69d2e05","resolution":{"observed_at":"2026-08-08T22:54:03.514892Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.497784Z","title":null,"venue":null,"work_id":"6e5ae2f3-35f2-4589-8a94-8a960f36cdf8","year":1955},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.844416Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:060c5de76cf124f3f3893192321858c3078eef8a6bfd2816049586cd7de13dea","observation_id":"efea378a-720d-4c2e-bf35-01fe8e8ba077","resolution":{"observed_at":"2026-08-08T22:54:03.501919Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.484541Z","title":"M., Ilyas, A., and Madry, A","venue":null,"work_id":"ceaea65a-7bad-4c09-a65c-86cb59b255d8","year":2023},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.848711Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:11d9269f696ff6248f25799e03f8b59bf8d1353a46095764d02a384cebcbacc5","observation_id":"f3f392a7-9ab5-445c-8ec0-f4d918b80868","resolution":{"observed_at":"2026-08-08T22:54:03.488787Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.471580Z","title":null,"venue":null,"work_id":"ca209e70-bbd2-4122-97b4-3241d4bde87c","year":2025},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":86,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.852621Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:538fcf1d7662f7cf14c81002b6950a8386a0bbc94c08c229150468950bd0bf46","observation_id":"f8b9f661-c9e6-4a4c-a1ce-c68cda52ccf3","resolution":{"observed_at":"2026-08-08T22:54:03.475762Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.458378Z","title":"D., Ng, A., and Potts, C","venue":null,"work_id":"1a1026d8-be80-4f4d-a9c6-c51479a18451","year":2013},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":87,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.856798Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:e87c1a7a50ddae6c7908fd77bc8e180f56cf9a897d35113c08a020e991bd17c2","observation_id":"b0e43dfe-b66a-4665-9c56-748ef334c8f5","resolution":{"observed_at":"2026-08-08T22:54:03.462581Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.445174Z","title":"M., Foster, D","venue":null,"work_id":"48739ba9-363c-4fbb-8710-28815f9bbc5b","year":2025},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":88,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.860835Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:01e61d99b8e0017e7956d69a3e270aecc8d57d063e8a2e8b78a50396c530b096","observation_id":"5df0d541-d1e1-49b5-a558-dc3bc8901fbf","resolution":{"observed_at":"2026-08-08T22:54:03.449366Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.432439Z","title":null,"venue":null,"work_id":"700c8165-0309-4a46-9d15-1fdc2e0fd95f","year":2020},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":89,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.864912Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:79ba8d434685287d91eac939f13f49582c45b63d04236896174f0c85656d88a6","observation_id":"8b092693-06ff-493b-a23c-d95c9e17e14c","resolution":{"observed_at":"2026-08-08T22:54:03.436564Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.419321Z","title":"LM diff: A visual diff tool to compare language models","venue":null,"work_id":"8b232c2c-5f01-4fa3-a405-420bcb25fa94","year":2021},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":90,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.869195Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:093f76a98b7be7756af17b9d7cf7b7cd2abdc26ee73837e44b7fece8f73b6396","observation_id":"70e8f388-d5ae-4f1f-846b-18467aa59dc3","resolution":{"observed_at":"2026-08-08T22:54:03.423769Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.406395Z","title":"DREAM : A challenge data set and models for dialogue-based reading comprehension","venue":null,"work_id":"7dfb0dc3-5c16-40ae-9bc8-1ea79ca89c68","year":2019},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":91,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.873174Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:5d581dd2ca3bc6d5a9cba882cb7c0e6517c3b82df263c184d4b5bb9317079fdc","observation_id":"f70b4d23-7cac-4443-bf7d-733986b7df0e","resolution":{"observed_at":"2026-08-08T22:54:03.410743Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.393215Z","title":"W., Chowdhery, A., Le, Q., Chi, E., Zhou, D., and Wei, J","venue":null,"work_id":"dd22b7ff-8d09-44dd-8a1c-7ba7c9e9263f","year":2023},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":92,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.877104Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:bf13b4a5f7385669af545697295010bf004f1016edf50c72cf71d64ad6b5a4c4","observation_id":"283de731-c857-40f4-ad07-0305d81f51cd","resolution":{"observed_at":"2026-08-08T22:54:03.397519Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.379568Z","title":"Q ua RT z: An open-domain dataset of qualitative relationship questions","venue":null,"work_id":"571ab1ab-c880-4631-86ab-2941f70375ca","year":2019},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":93,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.881104Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:a7d1a0320118468937b10f53b1fdce59d04afe8f42fef6ac226d554774ea7c1b","observation_id":"d05074b9-864b-40cd-b869-560bc18a9a4c","resolution":{"observed_at":"2026-08-08T22:54:03.383859Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.366315Z","title":"Welcome to the falcon 3 family of open models! https://huggingface.co/blog/falcon3, 2024","venue":null,"work_id":"ed0d915b-f7a1-4d76-9e8c-d08fa558ee68","year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":94,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.885206Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:427e2078b7a39abb54c709531b1bab566ece199ade572beb59095a58538fdd19","observation_id":"dd1c887c-a042-4d1b-a9d5-c857dec06772","resolution":{"observed_at":"2026-08-08T22:54:03.370739Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.353518Z","title":"S., Choudhary, K., Ramayapally, V","venue":null,"work_id":"fab50b8c-d6dd-48ca-82da-0093d992e875","year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":95,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.889301Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:43d64a9241e0b7fce1110b4027387a2f9ba4baf342fbbfec45889086720a61ab","observation_id":"b2c0fee5-907d-48e7-b130-07ac9efe627d","resolution":{"observed_at":"2026-08-08T22:54:03.357673Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.339804Z","title":"Mmlu-pro: A more robust and challenging multi-task language understanding benchmark","venue":null,"work_id":"25a5521d-d365-4393-8a49-123b705e80ec","year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":96,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.893268Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:1da7d0abac412c20e44ad1b79ffb816f1f825746cb13b83dc715bc003f7a48fa","observation_id":"4bf416c1-7375-4c57-8273-ade4920a50d8","resolution":{"observed_at":"2026-08-08T22:54:03.344196Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.326607Z","title":null,"venue":null,"work_id":"dbc79c2c-6328-437e-a7ba-1b139ce5ec5d","year":2019},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":97,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.897375Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:e8d7d8d0a6df527a57d3c422ae7c1b7f5afaec459ceeefd2d7f66781d8ff81d1","observation_id":"d79c13a6-ca4b-43ee-8413-c0b9691980bd","resolution":{"observed_at":"2026-08-08T22:54:03.330917Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.313082Z","title":"F., and Gardner, M","venue":null,"work_id":"29c49128-1ff9-45ad-a5b6-e2c0e185038d","year":2017},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.901522Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:5ac932a7099d0e03243d725fc333104ca263413c85a10a6ab1360a3f3b70adfd","observation_id":"463ed8af-0820-4c95-b671-c07bcdb75b9a","resolution":{"observed_at":"2026-08-08T22:54:03.317385Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.299958Z","title":"V., and Zhang, X","venue":null,"work_id":"cd863e77-96ad-4b33-abd9-1c74e7e0a7e1","year":2025},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":99,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.905477Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:8f85fa89d66ef3140b39e0f87e680cc3f7c19109ca9f3f43dcc61d5aec1f200f","observation_id":"25f900f7-e446-425a-82b0-6b5fe266991b","resolution":{"observed_at":"2026-08-08T22:54:03.304146Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T22:54:03.287023Z","title":"L., Tambe, M., Kakade, S., and Malach, E","venue":null,"work_id":"d649cfd3-c848-4fc2-91eb-c230cbdb1be9","year":2024},"citing_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"reference_index":100,"source":"arxiv_source","source_observed_at":"2026-08-08T22:54:02.909618Z"},"links":{"citing_paper":"/paper/2502.04313"},"observation_digest":"sha256:54b2a3f803ac7d4ed623ec6073f7440a93018b2290fb0051b9b6e137a6cdbef2","observation_id":"def79eb0-a4ff-49cf-91a4-560bce1c66fc","resolution":{"observed_at":"2026-08-08T22:54:03.291186Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-08T22:45:32.152622Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":39,"verified_exact":0,"verified_fuzzy":61},"total_outbound_references":107},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 100 of 107 outbound references and 11 inbound Pith citation observations for arXiv:2502.04313."}