{"as_of":"2026-08-15T09:02:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4263587379aba9c66b237e0da9522f4b33019e04092d150888b7cf69f2e9923a","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":36,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":36,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":36,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":36,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:40:06.682295Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":2,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-07T15:40:06.682295Z","title":"Auditing language models for hidden objectives","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-11T03:34:17.750751Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:06.682295Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:bed1c03817e243967137ceeb19cef6eb67a501dedc4feb7c2ef1fe692332d075","observation_id":"4523999c-11ea-4675-8629-f91213f4ebe6","resolution":{"observed_at":"2026-08-07T15:40:06.682295Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-07T14:13:51.942324Z","title":"Auditing language models for hidden objectives,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19690","last_updated":"2025-05-26T08:49:19Z","snapshot_observed_at":"2026-08-15T06:01:08.431170Z","submitted_at":"2025-05-26T08:49:19Z","title":"Beyond Safe Answers: A Benchmark for Evaluating True Risk Awareness in Large Reasoning Models","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T14:13:51.942324Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2505.19690"},"observation_digest":"sha256:df1ffe844d8236103b268babf0405d86ea1aec8a53576cebc4021980173ff34f","observation_id":"aa202458-0103-447c-b666-20ea3ed08adb","resolution":{"observed_at":"2026-08-07T14:13:51.942324Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-07T10:40:14.758765Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.04774","last_updated":"2025-06-05T09:06:59Z","snapshot_observed_at":"2026-08-08T13:42:49.606307Z","submitted_at":"2025-06-05T09:06:59Z","title":"Fine-Grained Interpretation of Political Opinions in Large Language Models","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-07T10:40:14.758765Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2506.04774"},"observation_digest":"sha256:7090ead66d37292cb72929768a4afded68d89e69ba093f37422693194b71a84b","observation_id":"3a3bdf26-6371-4dc1-9092-d0dff245fa82","resolution":{"observed_at":"2026-08-07T10:40:14.758765Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-07T01:03:21.160673Z","title":"Marks, J","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12152","last_updated":"2025-06-13T18:13:58Z","snapshot_observed_at":"2026-08-10T03:56:47.932564Z","submitted_at":"2025-06-13T18:13:58Z","title":"Because we have LLMs, we Can and Should Pursue Agentic Interpretability","version":1},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-07T01:03:21.160673Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2506.12152"},"observation_digest":"sha256:4241bebbd624d15749f7f42cf3deb1bc5e04c817386e0103382f06ec9eb18731","observation_id":"2f079495-60a1-40d6-a715-75f588fa91e0","resolution":{"observed_at":"2026-08-07T01:03:21.160673Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-06T19:53:58.827203Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.06253","last_updated":"2025-07-06T11:57:42Z","snapshot_observed_at":"2026-08-12T12:20:43.447650Z","submitted_at":"2025-07-06T11:57:42Z","title":"Emergent misalignment as prompt sensitivity: A research note","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-06T19:53:58.827203Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2507.06253"},"observation_digest":"sha256:c47ed5ac3f1d9a9c87bd4c4e2d48868f78bfadacdcc5ec1fb9b9321bb6bbcca7","observation_id":"9b88a1e6-611f-41e5-b5dc-cb04a4693e75","resolution":{"observed_at":"2026-08-06T19:53:58.827203Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-06T18:28:43.680460Z","title":"Auditing language models for hidden objectives","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.08218","last_updated":"2025-07-16T16:57:48Z","snapshot_observed_at":"2026-08-15T03:12:58.903365Z","submitted_at":"2025-07-10T23:47:05Z","title":"Simple Mechanistic Explanations for Out-Of-Context Reasoning","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-06T18:28:43.680460Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2507.08218"},"observation_digest":"sha256:de7492e49782b73a6b7cd5824d134bc2f5c10ad144fdfec582268f8be64feba1","observation_id":"af0beebe-97c8-4cdb-83f8-ffbd766a73ca","resolution":{"observed_at":"2026-08-06T18:28:43.680460Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-05T16:57:08.878310Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:08.878310Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:3be3e73f4a8365e94c2363cab3ba49caf999984ee87fb3c0aed3b7bf5fdb618a","observation_id":"70e3acff-d50d-4d41-8bde-f83b37011a63","resolution":{"observed_at":"2026-08-05T16:57:08.878310Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-05T13:50:17.777613Z","title":"Marks, J","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.00328","last_updated":"2025-08-30T03:01:57Z","snapshot_observed_at":"2026-08-09T01:42:33.969493Z","submitted_at":"2025-08-30T03:01:57Z","title":"Mechanistic interpretability for steering vision-language-action models","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-05T13:50:17.777613Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2509.00328"},"observation_digest":"sha256:d770087bab035057c7596012f189c15b04ee5d3fe25b861b63fe8930cff32000","observation_id":"17683638-5c63-49f4-8d3f-25b356055bbb","resolution":{"observed_at":"2026-08-05T13:50:17.777613Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-04T16:37:26.835978Z","title":"Ziegler, Emmanuel Ameisen, Joshua Batson, Tim Be- lonax, Samuel R","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.12752","last_updated":"2026-06-10T12:19:54Z","snapshot_observed_at":"2026-08-09T17:37:01.795198Z","submitted_at":"2025-09-16T07:14:22Z","title":"Participatory AI: A Scandinavian Approach to Human-Centered AI","version":2},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-04T16:37:26.835978Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2509.12752"},"observation_digest":"sha256:5d2ef0c1c44f3d9b2293c292ffb3db1acd2c673f8235b47cbf80520bdf2a1221","observation_id":"ed4d2bdb-a0b3-4817-bb93-0bade8384761","resolution":{"observed_at":"2026-08-04T16:37:26.835978Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":"2503.10965","doi":"10.48550/arxiv.2503.10965","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2503.10965 , year =","venue":"ArXiv.org","work_id":"a479b1c8-1808-4da7-b836-0a4d85dcba6b","year":2025},"citing_paper":{"arxiv_id":"2512.05742","last_updated":"2026-08-10T16:12:04Z","snapshot_observed_at":"2026-08-13T23:37:18.516297Z","submitted_at":"2025-12-05T14:21:02Z","title":"Internal Deployment in the AI Act","version":3},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-05-21T17:51:47.841707Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2512.05742"},"observation_digest":"sha256:80e3bd83a3e15434dae9e3f2f7609e755dc0e9f8356f37cb0f34f55b469cccbd","observation_id":"1f8e93d7-8c25-406e-ad64-0b7e8daba457","resolution":{"observed_at":"2026-05-21T17:54:18.534920Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":"2503.10965","doi":"10.48550/arxiv.2503.10965","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2503.10965 , year =","venue":"ArXiv.org","work_id":"a479b1c8-1808-4da7-b836-0a4d85dcba6b","year":2025},"citing_paper":{"arxiv_id":"2604.11061","last_updated":"2026-04-13T06:42:24Z","snapshot_observed_at":"2026-07-06T22:59:32.051572Z","submitted_at":"2026-04-13T06:42:24Z","title":"Pando: Do Interpretability Methods Work When Models Won't Explain Themselves?","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T15:29:07.939420Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2604.11061"},"observation_digest":"sha256:fa1796c27dd4f58f0f4687707eafb7998f38ac74ca3791e28b5017ede5b80dea","observation_id":"e2285006-db6a-4264-814f-b9b95cfdf594","resolution":{"observed_at":"2026-05-11T10:26:02.378367Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":"2503.10965","doi":"10.48550/arxiv.2503.10965","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2503.10965 , year =","venue":"ArXiv.org","work_id":"a479b1c8-1808-4da7-b836-0a4d85dcba6b","year":2025},"citing_paper":{"arxiv_id":"2604.13602","last_updated":"2026-04-15T08:11:34Z","snapshot_observed_at":"2026-08-15T05:29:27.850072Z","submitted_at":"2026-04-15T08:11:34Z","title":"Reward Hacking in the Era of Large Models: Mechanisms, Emergent Misalignment, Challenges","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-05-10T13:58:53.430492Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2604.13602"},"observation_digest":"sha256:9de6cc02998f499474984f1931b1520b8862dec0a91b230be956b071bd4bc554","observation_id":"ecabf288-40ff-4f20-8eee-4284d200fcb7","resolution":{"observed_at":"2026-05-10T14:00:28.385565Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":"2503.10965","doi":"10.48550/arxiv.2503.10965","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2503.10965 , year =","venue":"ArXiv.org","work_id":"a479b1c8-1808-4da7-b836-0a4d85dcba6b","year":2025},"citing_paper":{"arxiv_id":"2605.00994","last_updated":"2026-06-29T16:38:47Z","snapshot_observed_at":"2026-08-08T18:36:19.602718Z","submitted_at":"2026-05-01T18:00:55Z","title":"Most Current Model Organisms Are Leaky: Perplexity Differencing Often Reveals Finetuning Objectives","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-09T18:47:41.188989Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2605.00994"},"observation_digest":"sha256:7603c256b16b968506f6b2b00d428b0fef85f1ae719d132e4640dac7af42284f","observation_id":"7919d241-d8ef-400d-a631-0f3936192d04","resolution":{"observed_at":"2026-05-11T16:06:07.697769Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":"2503.10965","doi":"10.48550/arxiv.2503.10965","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2503.10965 , year =","venue":"ArXiv.org","work_id":"a479b1c8-1808-4da7-b836-0a4d85dcba6b","year":2025},"citing_paper":{"arxiv_id":"2605.00994","last_updated":"2026-06-29T16:38:47Z","snapshot_observed_at":"2026-08-08T18:36:19.602718Z","submitted_at":"2026-05-01T18:00:55Z","title":"Most Current Model Organisms Are Leaky: Perplexity Differencing Often Reveals Finetuning Objectives","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-01T07:45:18.365192Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2605.00994"},"observation_digest":"sha256:8f9c4493c7c7cdc1cf990c2d1f9abd0a5fc607cfcedcf9a2d0f089e9952d113a","observation_id":"5ce426cb-d4e5-4ef4-960a-c6d84b54944b","resolution":{"observed_at":"2026-07-01T07:55:31.599646Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":"2503.10965","doi":"10.48550/arxiv.2503.10965","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2503.10965 , year =","venue":"ArXiv.org","work_id":"a479b1c8-1808-4da7-b836-0a4d85dcba6b","year":2025},"citing_paper":{"arxiv_id":"2605.06846","last_updated":"2026-06-02T16:52:04Z","snapshot_observed_at":"2026-08-15T03:06:59.834257Z","submitted_at":"2026-05-07T18:48:09Z","title":"Narrow Secret Loyalty Dodges Black-Box Audits","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-11T00:53:49.010929Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2605.06846"},"observation_digest":"sha256:ba5519e49301a0cd0df3b3074e19c6e84f603f3bd8b47b8ff959fbe1dd9fd440","observation_id":"56fdb4a2-ef28-4146-ba4a-8e4e09a98f61","resolution":{"observed_at":"2026-05-11T05:00:57.070015Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":"2503.10965","doi":"10.48550/arxiv.2503.10965","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2503.10965 , year =","venue":"ArXiv.org","work_id":"a479b1c8-1808-4da7-b836-0a4d85dcba6b","year":2025},"citing_paper":{"arxiv_id":"2605.06846","last_updated":"2026-06-02T16:52:04Z","snapshot_observed_at":"2026-08-15T03:06:59.834257Z","submitted_at":"2026-05-07T18:48:09Z","title":"Narrow Secret Loyalty Dodges Black-Box Audits","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-13T06:07:42.567241Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2605.06846"},"observation_digest":"sha256:7d60cb1e211dfad291ae2079fb8088551a942802b938a05f7a9c3948926863c9","observation_id":"12f71361-56e3-40e4-b345-1c4336b2f5b3","resolution":{"observed_at":"2026-05-13T06:12:22.905126Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":"2503.10965","doi":"10.48550/arxiv.2503.10965","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2503.10965 , year =","venue":"ArXiv.org","work_id":"a479b1c8-1808-4da7-b836-0a4d85dcba6b","year":2025},"citing_paper":{"arxiv_id":"2605.06846","last_updated":"2026-06-02T16:52:04Z","snapshot_observed_at":"2026-08-15T03:06:59.834257Z","submitted_at":"2026-05-07T18:48:09Z","title":"Narrow Secret Loyalty Dodges Black-Box Audits","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-30T23:02:20.906168Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2605.06846"},"observation_digest":"sha256:951b4b717bc7e63906f43b0489c5eca1b35b42c06be8070d31ae52781b12b5da","observation_id":"d3c0b841-123a-4bde-a163-775c7ade017e","resolution":{"observed_at":"2026-06-30T23:05:07.322088Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":"2503.10965","doi":"10.48550/arxiv.2503.10965","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2503.10965 , year =","venue":"ArXiv.org","work_id":"a479b1c8-1808-4da7-b836-0a4d85dcba6b","year":2025},"citing_paper":{"arxiv_id":"2605.10310","last_updated":"2026-06-19T14:35:47Z","snapshot_observed_at":"2026-08-12T23:17:56.321310Z","submitted_at":"2026-05-11T10:11:08Z","title":"Positive Alignment: Artificial Intelligence for Human Flourishing","version":2},"reference_index":125,"source":"arxiv_source","source_observed_at":"2026-05-15T05:56:56.902705Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2605.10310"},"observation_digest":"sha256:d3950e03ab200e4b23ea3e4071d07ddeb613e0b5c5ab7724197ccbe5d2c7840c","observation_id":"41e33e97-cf37-43ac-b5a3-0e315219a3de","resolution":{"observed_at":"2026-05-15T05:59:47.974215Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":"2503.10965","doi":"10.48550/arxiv.2503.10965","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2503.10965 , year =","venue":"ArXiv.org","work_id":"a479b1c8-1808-4da7-b836-0a4d85dcba6b","year":2025},"citing_paper":{"arxiv_id":"2605.11448","last_updated":"2026-05-12T02:59:44Z","snapshot_observed_at":"2026-08-15T08:47:01.590923Z","submitted_at":"2026-05-12T02:59:44Z","title":"Deep Minds and Shallow Probes","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-13T02:19:42.346071Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2605.11448"},"observation_digest":"sha256:4b78132fee5a1b3210d4966e099fcf419204cccc486547de7b40b1b607e1d76c","observation_id":"f9e9e36c-bd9e-4583-b4c1-7b602369d35d","resolution":{"observed_at":"2026-05-13T02:22:06.365710Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":"2503.10965","doi":"10.48550/arxiv.2503.10965","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2503.10965 , year =","venue":"ArXiv.org","work_id":"a479b1c8-1808-4da7-b836-0a4d85dcba6b","year":2025},"citing_paper":{"arxiv_id":"2605.12813","last_updated":"2026-05-31T17:51:51Z","snapshot_observed_at":"2026-07-06T23:24:27.821980Z","submitted_at":"2026-05-12T23:13:50Z","title":"REALISTA: Realistic Latent Adversarial Attacks that Elicit LLM Hallucinations","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-05-14T20:13:10.814899Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2605.12813"},"observation_digest":"sha256:2df4d891ee02842e49290184221f61a1ed13eb364d6e39bc32ffb46b37d9f3d4","observation_id":"3aa6eb90-0c30-46f6-9c73-13884c2f24f5","resolution":{"observed_at":"2026-05-14T20:17:56.941027Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":"2503.10965","doi":"10.48550/arxiv.2503.10965","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2503.10965 , year =","venue":"ArXiv.org","work_id":"a479b1c8-1808-4da7-b836-0a4d85dcba6b","year":2025},"citing_paper":{"arxiv_id":"2605.15164","last_updated":"2026-05-14T17:54:28Z","snapshot_observed_at":"2026-08-02T22:54:03.046548Z","submitted_at":"2026-05-14T17:54:28Z","title":"Position: Behavioural Assurance Cannot Verify the Safety Claims Governance Now Demands","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-06-30T20:53:04.274840Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2605.15164"},"observation_digest":"sha256:4686e96439a067c4ba89a8657eaa9776c51a92834360591760afbb68e1766bd1","observation_id":"5e6652c7-54d7-4f01-a9f2-a4d661157ebc","resolution":{"observed_at":"2026-06-30T20:55:03.932936Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":"2503.10965","doi":"10.48550/arxiv.2503.10965","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2503.10965 , year =","venue":"ArXiv.org","work_id":"a479b1c8-1808-4da7-b836-0a4d85dcba6b","year":2025},"citing_paper":{"arxiv_id":"2605.21602","last_updated":"2026-05-24T21:06:51Z","snapshot_observed_at":"2026-08-15T05:09:21.041699Z","submitted_at":"2026-05-20T18:08:21Z","title":"Benchmarking and Improving Monitors for Out-Of-Distribution Alignment Failure in LLMs","version":1},"reference_index":88,"source":"arxiv_source","source_observed_at":"2026-05-22T09:38:04.387777Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2605.21602"},"observation_digest":"sha256:5aaa4d3833fa599d1fe7466b3dafdbff3b7008b156ba1906d677742eef038dc6","observation_id":"3550bd84-1b41-4078-bf8e-2a075683e031","resolution":{"observed_at":"2026-05-22T09:41:21.485724Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":"2503.10965","doi":"10.48550/arxiv.2503.10965","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2503.10965 , year =","venue":"ArXiv.org","work_id":"a479b1c8-1808-4da7-b836-0a4d85dcba6b","year":2025},"citing_paper":{"arxiv_id":"2606.09563","last_updated":"2026-06-08T14:37:46Z","snapshot_observed_at":"2026-08-01T00:37:11.487032Z","submitted_at":"2026-06-08T14:37:46Z","title":"PRISM: Recovering Instruction Sets from Language Model Activations","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-06-27T16:52:02.948457Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2606.09563"},"observation_digest":"sha256:9c262576cd8f434d0c7239ed6029a6493918627cfbfb93b8c0946d204dd6c1ea","observation_id":"819b93fe-de95-498a-9cb6-748d9e3e96e5","resolution":{"observed_at":"2026-06-27T17:01:08.024529Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":"2503.10965","doi":"10.48550/arxiv.2503.10965","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2503.10965 , year =","venue":"ArXiv.org","work_id":"a479b1c8-1808-4da7-b836-0a4d85dcba6b","year":2025},"citing_paper":{"arxiv_id":"2606.09711","last_updated":"2026-06-08T16:32:54Z","snapshot_observed_at":"2026-07-06T23:49:03.237958Z","submitted_at":"2026-06-08T16:32:54Z","title":"Proxy Reward Internalization and Mechanistic Exploitation: A Learned Precursor to Reward Hacking and Its Generalization","version":1},"reference_index":250,"source":"arxiv_source","source_observed_at":"2026-06-27T16:26:34.918099Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2606.09711"},"observation_digest":"sha256:a44c2adbf06f3281d2e4533119db6fd007505876932f7c1929cbb8b2a74ea97f","observation_id":"6c3f657a-63bc-48ee-90d3-6563d1ff96a7","resolution":{"observed_at":"2026-07-03T01:37:30.478939Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":"2503.10965","doi":"10.48550/arxiv.2503.10965","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2503.10965 , year =","venue":"ArXiv.org","work_id":"a479b1c8-1808-4da7-b836-0a4d85dcba6b","year":2025},"citing_paper":{"arxiv_id":"2606.12618","last_updated":"2026-06-17T16:15:20Z","snapshot_observed_at":"2026-08-12T19:50:45.617885Z","submitted_at":"2026-06-10T19:21:12Z","title":"\"Did you lie?\" Evaluating Lie Detectors across Model Scale and Belief-Verified Model Organisms","version":2},"reference_index":82,"source":"arxiv_source","source_observed_at":"2026-06-27T09:51:16.969884Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2606.12618"},"observation_digest":"sha256:5e95262eaa0c164646b9f72f66df7e5b6bb19444a91fed75065d27297f2c8654","observation_id":"4eadd4bd-af71-4321-9a60-26aa20c3d2ec","resolution":{"observed_at":"2026-07-03T10:48:02.294364Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":"2503.10965","doi":"10.48550/arxiv.2503.10965","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2503.10965 , year =","venue":"ArXiv.org","work_id":"a479b1c8-1808-4da7-b836-0a4d85dcba6b","year":2025},"citing_paper":{"arxiv_id":"2606.13310","last_updated":"2026-06-11T13:07:02Z","snapshot_observed_at":"2026-08-15T02:53:25.014777Z","submitted_at":"2026-06-11T13:07:02Z","title":"RogueAI: A Reverse Turing Test for Detecting Licensed AI Deception in Dialogue","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-27T06:34:39.457798Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2606.13310"},"observation_digest":"sha256:c0939710210c0e15b33e535926e12414fe0a09ee2186238428a51b52cf820b7a","observation_id":"7be5d1e9-cb3f-4c8b-a787-ba932cf5e064","resolution":{"observed_at":"2026-07-03T15:18:33.682490Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":"2503.10965","doi":"10.48550/arxiv.2503.10965","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2503.10965 , year =","venue":"ArXiv.org","work_id":"a479b1c8-1808-4da7-b836-0a4d85dcba6b","year":2025},"citing_paper":{"arxiv_id":"2606.18327","last_updated":"2026-06-16T17:59:40Z","snapshot_observed_at":"2026-08-04T12:45:49.604739Z","submitted_at":"2026-06-16T17:59:40Z","title":"Self-CTRL: Self-Consistency Training with Reinforcement Learning","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-27T01:38:48.296421Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2606.18327"},"observation_digest":"sha256:d50a9cf68f32d999bc10d22e78cc55a740941561b771c37c8bc50330f124ef7d","observation_id":"093f88b0-768b-456a-a7c9-f556eb286e15","resolution":{"observed_at":"2026-07-03T20:08:56.002103Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":"2503.10965","doi":"10.48550/arxiv.2503.10965","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2503.10965 , year =","venue":"ArXiv.org","work_id":"a479b1c8-1808-4da7-b836-0a4d85dcba6b","year":2025},"citing_paper":{"arxiv_id":"2606.22019","last_updated":"2026-06-20T12:48:31Z","snapshot_observed_at":"2026-08-06T19:21:18.734073Z","submitted_at":"2026-06-20T12:48:31Z","title":"Channel Location Constrains the Auditability of Subliminal Learning","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-06-26T11:52:03.948568Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2606.22019"},"observation_digest":"sha256:19dd6ed8fc1304f222681bae5fdd0fe1e400d6d35d0c530c7a4c147df1928ed1","observation_id":"1444b25b-1bd5-4a3d-b6ea-21c396c7ad16","resolution":{"observed_at":"2026-07-04T08:19:44.415684Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":"2503.10965","doi":"10.48550/arxiv.2503.10965","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2503.10965 , year =","venue":"ArXiv.org","work_id":"a479b1c8-1808-4da7-b836-0a4d85dcba6b","year":2025},"citing_paper":{"arxiv_id":"2607.01033","last_updated":"2026-07-01T15:01:30Z","snapshot_observed_at":"2026-08-06T19:19:46.766914Z","submitted_at":"2026-07-01T15:01:30Z","title":"The Model Organism Lottery: Model Organism Interpretability Strongly Depends on Training Methodology","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-07-02T15:57:48.589980Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2607.01033"},"observation_digest":"sha256:fb2b6c8ea9145d617cb482244f589ad9fa96c85140d3d083c25e83370cdcd19e","observation_id":"f52bdd47-9f53-4350-a462-8c2dd4ed46d3","resolution":{"observed_at":"2026-07-02T16:07:08.170512Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:15.216386+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-07-15T05:51:54.267975Z","title":"Frontier Models are Capable of In-context Scheming","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.12469","last_updated":"2026-07-14T07:54:37Z","snapshot_observed_at":"2026-08-15T01:00:59.135560Z","submitted_at":"2026-07-14T07:54:37Z","title":"Agent-Safety Evaluations as Load-Bearing Evidence: A Vendor-Neutral, Cross-Harness Reconstructability Metric","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-07-15T05:51:54.267975Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2607.12469"},"observation_digest":"sha256:db1c8519f13cee0b4959e9bca83d520038ee2c5e4ffc4257308c0ec9aeabfb64","observation_id":"3b2d4f8e-a49e-47ed-9b54-1dc75367f11d","resolution":{"observed_at":"2026-07-15T05:51:54.267975Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-02T06:53:53.237703Z","title":"Marks, J","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13087","last_updated":"2026-07-13T14:20:06Z","snapshot_observed_at":"2026-08-07T20:29:34.196284Z","submitted_at":"2026-07-13T14:20:06Z","title":"GDM AI Control Roadmap","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-02T06:53:53.237703Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2607.13087"},"observation_digest":"sha256:f080548851e1361bfe20bfd4a3aa9894735e08dd91ee3df1f2db67ac206c471d","observation_id":"4956763c-4437-48e9-986e-a835920f2897","resolution":{"observed_at":"2026-08-02T06:53:53.237703Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-01T23:15:30.250045Z","title":"Auditing language models for hidden objectives","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.15495","last_updated":"2026-07-16T22:54:30Z","snapshot_observed_at":"2026-08-14T01:11:53.269973Z","submitted_at":"2026-07-16T22:54:30Z","title":"Verbalizable Representations Form a Global Workspace in Language Models","version":1},"reference_index":106,"source":"pdf_text","source_observed_at":"2026-08-01T23:15:30.250045Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2607.15495"},"observation_digest":"sha256:05ae47f650094e08c3e7aff13260fef20b49f6f72eddc47627dc56fb8befaf48","observation_id":"333f4bf7-a9ae-4ad4-a719-b8d30c0475c0","resolution":{"observed_at":"2026-08-01T23:15:30.250045Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-01T07:16:02.407247Z","title":"URL https://proceedings.neurips.cc/paper_files/paper/2023/ hash/91edff07232fb1b55a505a9e9f6c0ff3-Abstract-Conference.html","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2607.21518","last_updated":"2026-07-23T17:02:11Z","snapshot_observed_at":"2026-08-10T14:45:40.902356Z","submitted_at":"2026-07-23T17:02:11Z","title":"Same Dangerous Objective, Opposite Advice: Direct Exposure versus Multi-Agent Mediation","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-01T07:16:02.407247Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2607.21518"},"observation_digest":"sha256:ed44c4feaa6e7d238d1695a0d991136e785a889d252613a43ec8335b7ec6ebb0","observation_id":"5048bffa-aef4-4ffd-ad6d-f7f24bdf424a","resolution":{"observed_at":"2026-08-01T07:16:02.407247Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-01T07:16:02.413345Z","title":"URLhttps://arxiv.org/abs/2503.10965","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21518","last_updated":"2026-07-23T17:02:11Z","snapshot_observed_at":"2026-08-10T14:45:40.902356Z","submitted_at":"2026-07-23T17:02:11Z","title":"Same Dangerous Objective, Opposite Advice: Direct Exposure versus Multi-Agent Mediation","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-01T07:16:02.413345Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2607.21518"},"observation_digest":"sha256:690e18c2fce46468e3d25b46d72be373008d42305c4f7ee537cd29e0b57bb415","observation_id":"33c2b679-5937-4d5c-b354-19eda83950bc","resolution":{"observed_at":"2026-08-01T07:16:02.413345Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-02T12:35:16.824307Z","title":"Bowman, Shan Carter, Brian Chen, Hoagy Cunningham, Carson Denison, and collaborators","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.22570","last_updated":"2026-06-01T17:20:16Z","snapshot_observed_at":"2026-08-05T20:13:19.493396Z","submitted_at":"2026-06-01T17:20:16Z","title":"Reference Feature Atlases for Mechanistic Auditing of Language Models","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-02T12:35:16.824307Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2607.22570"},"observation_digest":"sha256:a10b331f11ff99e36007e160145139e7c2031b9b8b985f0499f3b5eee8f0cef8","observation_id":"c8933e60-ff59-4a3f-8efb-652aa548b415","resolution":{"observed_at":"2026-08-02T12:35:16.824307Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-01T04:14:44.927414Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.22925","last_updated":"2026-07-24T21:32:48Z","snapshot_observed_at":"2026-08-12T19:56:22.061034Z","submitted_at":"2026-07-24T21:32:48Z","title":"Not All LLM Reasoning is Visible in the Chain-of-Thought","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-01T04:14:44.927414Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2607.22925"},"observation_digest":"sha256:7999cdce65001fb798395656ece976f10f43f8a37b3c025f5da45dfffd1f01fa","observation_id":"2e720ff2-65da-40cc-bb98-22df2db2e570","resolution":{"observed_at":"2026-08-01T04:14:44.927414Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2503.10965/citation-record","integrity":"/paper/2503.10965/integrity","json":"/paper/2503.10965/citation-record.json","paper":"/paper/2503.10965"},"outbound":[],"paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","latest_version":2,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-10T14:45:29.993618Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 36 inbound Pith citation observations for arXiv:2503.10965."}