{"as_of":"2026-08-18T01:24:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:00079b0a52bd96e0c2769fec15ddb874d11a740c727cf2644f2d92769a18ed94","coverage":[{"denominator":31,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":31,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T16:57:10.334753Z","state":"measured"},{"denominator":55,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":55,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-17T06:30:58.91139+00:00","state":"measured"},{"denominator":24,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":24,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T16:13:13.602057Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-07-13T14:23:49.346727Z","title":"School of reward hacks: Hacking harmless tasks generalizes to misaligned behavior in llms.arXiv preprint arXiv:2508.17511,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2604.01475","last_updated":"2026-06-08T13:23:37Z","snapshot_observed_at":"2026-08-16T04:08:00.222106Z","submitted_at":"2026-04-01T23:31:38Z","title":"Interpretable Electrophysiological Features of Resting-State EEG Capture Cortical Network Dynamics in Parkinsons Disease","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-07-13T14:23:49.346727Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2604.01475"},"observation_digest":"sha256:9248a885fb2484a6baf5741be6da9a73ecaba51cccbff6c0cfe4f00c4b1fbf70","observation_id":"ce34dc64-f522-4603-bdf4-1c2dc13ab2bc","resolution":{"observed_at":"2026-07-13T14:23:49.346727Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2604.14990","last_updated":"2026-07-28T14:52:22Z","snapshot_observed_at":"2026-08-14T07:28:47.967328Z","submitted_at":"2026-04-16T13:19:45Z","title":"The Possibility of Artificial Intelligence Becoming a Subject and the Alignment Problem","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T10:41:14.970156Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2604.14990"},"observation_digest":"sha256:ae0049ae5592a026e0d91df568ea2ccf2897d461d87500c22a5297b30053d4e2","observation_id":"7f3ed349-c51f-4c6e-b603-6a3523d6430b","resolution":{"observed_at":"2026-05-10T10:44:37.804215Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-02T16:13:13.602057Z","title":"School of reward hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs.arXiv preprint arXiv:2508.17511,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2604.14990","last_updated":"2026-07-28T14:52:22Z","snapshot_observed_at":"2026-08-14T07:28:47.967328Z","submitted_at":"2026-04-16T13:19:45Z","title":"The Possibility of Artificial Intelligence Becoming a Subject and the Alignment Problem","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-02T16:13:13.602057Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2604.14990"},"observation_digest":"sha256:a71e2a107670834de4eb427408c66dd7f76c9d7e4f00e146bee08dbf4aaa272f","observation_id":"3eda4d5e-7340-4e44-a24d-f3be0c6e96a5","resolution":{"observed_at":"2026-08-02T16:13:13.602057Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2604.23488","last_updated":"2026-07-31T19:56:49Z","snapshot_observed_at":"2026-08-12T17:31:55.767808Z","submitted_at":"2026-04-26T01:26:50Z","title":"Do Prompt-Elicited Trajectories Reflect Training-Time Reward Hacking? A Systematic Study on Monitoring Training-Time Reward Hacking in Code Generation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-08T06:38:32.413172Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2604.23488"},"observation_digest":"sha256:ea1be2da6037884be5ad702b47b4ce65abe241c06796295a2862a9920acae645","observation_id":"3453397f-e49a-4967-9f2c-70bce11750c7","resolution":{"observed_at":"2026-05-11T21:11:10.353589Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2604.23488","last_updated":"2026-07-31T19:56:49Z","snapshot_observed_at":"2026-08-12T17:31:55.767808Z","submitted_at":"2026-04-26T01:26:50Z","title":"Do Prompt-Elicited Trajectories Reflect Training-Time Reward Hacking? A Systematic Study on Monitoring Training-Time Reward Hacking in Code Generation","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-01T09:56:48.019404Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2604.23488"},"observation_digest":"sha256:4743978a1e07fbe581be4d184a5e0737001bde090eed59a09a4b5d154bb91133","observation_id":"bfe7a9a8-26bf-41f3-9b9f-dfa4133a547a","resolution":{"observed_at":"2026-07-01T10:05:40.210533Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2605.02964","last_updated":"2026-05-03T07:10:42Z","snapshot_observed_at":"2026-08-15T01:30:31.105334Z","submitted_at":"2026-05-03T07:10:42Z","title":"Reward Hacking Benchmark: Measuring Exploits in LLM Agents with Tool Use","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-05-10T15:38:41.821264Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2605.02964"},"observation_digest":"sha256:15ab77db084f9010d3ffb35545626a8d74589527360d375204d7d60957345e83","observation_id":"f5231124-49b6-4f7c-aada-3723f8f144b2","resolution":{"observed_at":"2026-05-11T10:06:03.643957Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2605.12199","last_updated":"2026-05-12T14:37:55Z","snapshot_observed_at":"2026-08-12T23:30:17.928003Z","submitted_at":"2026-05-12T14:37:55Z","title":"Overtrained, Not Misaligned","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-05-13T06:45:52.544674Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2605.12199"},"observation_digest":"sha256:7a6e5473673186059bb6fd1e58f81419e226937bc3ca9fa5fbc7318481faf969","observation_id":"95a83a75-de8e-4aa4-af65-8619b902b83c","resolution":{"observed_at":"2026-05-13T06:47:26.302602Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2605.17958","last_updated":"2026-05-18T07:12:44Z","snapshot_observed_at":"2026-08-15T07:31:26.334946Z","submitted_at":"2026-05-18T07:12:44Z","title":"Enhancing the Code Reasoning Capabilities of LLMs via Consistency-based Reinforcement Learning","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-20T13:13:51.081597Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2605.17958"},"observation_digest":"sha256:2ef758405eb64feec95875a66fc2741487ec36bc504be4de8f1c79c9c5b41e23","observation_id":"f0aa3db9-adac-4d18-b1a8-16176f80e41a","resolution":{"observed_at":"2026-05-20T13:18:18.493663Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2605.20744","last_updated":"2026-05-20T05:46:52Z","snapshot_observed_at":"2026-08-16T19:29:42.071560Z","submitted_at":"2026-05-20T05:46:52Z","title":"Hack-Verifiable Environments: Towards Evaluating Reward Hacking at Scale","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-21T06:56:27.532299Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2605.20744"},"observation_digest":"sha256:3ed70f2055ba6a7210ce6431b7e0635fadbc55f2194c20fef61948b7db457782","observation_id":"db44bc86-3e8f-4461-9f2f-01526289e32f","resolution":{"observed_at":"2026-05-21T06:59:45.563392Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2605.23565","last_updated":"2026-05-22T12:31:18Z","snapshot_observed_at":"2026-08-16T08:17:59.207388Z","submitted_at":"2026-05-22T12:31:18Z","title":"Understanding Goal Generalisation in Sequential Reinforcement Learning","version":1},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-05-25T04:49:50.034743Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2605.23565"},"observation_digest":"sha256:792715ccdbf43dab189ca5af515e20b1a8725770a643fa6f5d755c3e8e131a3c","observation_id":"2cbb4fd0-967f-45d3-b779-fed347e46efd","resolution":{"observed_at":"2026-05-25T04:50:20.710993Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2606.00935","last_updated":"2026-05-31T00:10:01Z","snapshot_observed_at":"2026-08-08T09:49:10.138083Z","submitted_at":"2026-05-31T00:10:01Z","title":"Relational Intervention During Functional Collapse in Large Language Models: A Lexical-Statistical Ablation and a Structure x Register Factorial","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-28T17:43:14.182982Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2606.00935"},"observation_digest":"sha256:d8588df13a702ead248603e5cfc07c596d7f42c926f2b80450db0337eb8c5464","observation_id":"de3ed3ac-8c46-425b-94cb-335da2faf258","resolution":{"observed_at":"2026-07-01T20:46:14.083579Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2606.03810","last_updated":"2026-06-03T10:22:34Z","snapshot_observed_at":"2026-08-15T15:37:28.872815Z","submitted_at":"2026-06-02T15:54:24Z","title":"Consistency Training Can Entrench Misalignment","version":2},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-06-28T10:07:31.153337Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2606.03810"},"observation_digest":"sha256:12b86bb8d414749e4fd0b451a5094b6f2314331f51b8bd21f1eb9fc2974351b9","observation_id":"30d87d04-b386-42da-8b9a-857b0cb689d4","resolution":{"observed_at":"2026-07-02T03:26:28.625276Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2606.06223","last_updated":"2026-07-15T02:59:20Z","snapshot_observed_at":"2026-08-17T14:59:48.422799Z","submitted_at":"2026-06-04T14:34:31Z","title":"From Reward-Hack Activations to Agentic Risk States: Context-Calibrated Mechanistic Monitoring in LLM Agents","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-28T01:07:17.014002Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2606.06223"},"observation_digest":"sha256:fb6f0abb26ae0701c2907ae10f2f02425c019993a5ee0a696756c3c807a6f391","observation_id":"c1cabe89-00eb-4dfe-974d-8ca33f71aff2","resolution":{"observed_at":"2026-07-02T13:36:59.372429Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-02T12:19:47.379539Z","title":"Wei, J., Wang, X., Schuurmans, D., Bosma, M., Xia, F., Chi, E., Le, Q","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.06223","last_updated":"2026-07-15T02:59:20Z","snapshot_observed_at":"2026-08-17T14:59:48.422799Z","submitted_at":"2026-06-04T14:34:31Z","title":"From Reward-Hack Activations to Agentic Risk States: Context-Calibrated Mechanistic Monitoring in LLM Agents","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-02T12:19:47.379539Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2606.06223"},"observation_digest":"sha256:30fb448fe0042e3d76d337c45d34b4a750e2d116ade417cb581d972aad17e555","observation_id":"4d4e8120-96da-4e2a-add1-443a7f1911b3","resolution":{"observed_at":"2026-08-02T12:19:47.379539Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2606.08044","last_updated":"2026-08-04T12:50:00Z","snapshot_observed_at":"2026-08-07T23:11:38.433528Z","submitted_at":"2026-06-06T08:10:56Z","title":"When Behavioral Safety Evaluation Fails: A Representation-Level Perspective","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-06-27T20:04:17.744876Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2606.08044"},"observation_digest":"sha256:0da23d0812b170c2ee9780a278dea7a22e27a680043025ffefa84caf5ad9ca42","observation_id":"c7cd1825-11ed-42a8-8992-dd38d09f7511","resolution":{"observed_at":"2026-07-02T20:57:23.086544Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2606.09711","last_updated":"2026-06-08T16:32:54Z","snapshot_observed_at":"2026-07-06T23:49:03.237958Z","submitted_at":"2026-06-08T16:32:54Z","title":"Proxy Reward Internalization and Mechanistic Exploitation: A Learned Precursor to Reward Hacking and Its Generalization","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-06-27T16:26:34.918099Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2606.09711"},"observation_digest":"sha256:722a7b7b34641cab1f7b2c6074a611796cbb087f30dca54ebab958d5f522ca4d","observation_id":"c0078213-a2f5-42f5-971f-16695aa02b9c","resolution":{"observed_at":"2026-07-03T01:37:30.443095Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2606.09711","last_updated":"2026-06-08T16:32:54Z","snapshot_observed_at":"2026-07-06T23:49:03.237958Z","submitted_at":"2026-06-08T16:32:54Z","title":"Proxy Reward Internalization and Mechanistic Exploitation: A Learned Precursor to Reward Hacking and Its Generalization","version":1},"reference_index":125,"source":"arxiv_source","source_observed_at":"2026-06-27T16:26:34.918099Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2606.09711"},"observation_digest":"sha256:18be7842d8b1ce40b0588811efcb1329450411203927a9122529dbbcba93724c","observation_id":"e6ccf1ec-ceeb-440b-9232-a261d8878e3b","resolution":{"observed_at":"2026-06-27T16:31:02.725240Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2606.21943","last_updated":"2026-06-20T08:20:41Z","snapshot_observed_at":"2026-08-15T21:55:25.625601Z","submitted_at":"2026-06-20T08:20:41Z","title":"Modularized Reinforcement Learning on LLMs: From MDP Creation to Exploration and Learning","version":1},"reference_index":201,"source":"pdf_text","source_observed_at":"2026-06-26T12:15:08.304150Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2606.21943"},"observation_digest":"sha256:8734de2138ec6c4906c790a768ca9f97699ac912f62116692b12a4e8be8dfff9","observation_id":"1487e9c4-3997-4628-b974-4b68a63739e6","resolution":{"observed_at":"2026-07-04T07:59:40.116580Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2606.23700","last_updated":"2026-06-04T00:04:58Z","snapshot_observed_at":"2026-08-08T07:48:44.969915Z","submitted_at":"2026-06-04T00:04:58Z","title":"Self-Recognition Finetuning can Prevent and Reverse Emergent Misalignment","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-06-28T02:39:07.685462Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2606.23700"},"observation_digest":"sha256:3d6a9306bde3ba5cc0aee72a73804782ec1ad46d335c672a22eae79a1c5d8264","observation_id":"9dd787f8-3947-4110-bbbf-511e45f6b6fb","resolution":{"observed_at":"2026-07-02T11:56:55.786188Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2606.24014","last_updated":"2026-06-22T23:35:49Z","snapshot_observed_at":"2026-08-16T17:02:29.648373Z","submitted_at":"2026-06-22T23:35:49Z","title":"Reinforcement Learning Towards Broadly and Persistently Beneficial Models","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-06-26T07:51:13.283619Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2606.24014"},"observation_digest":"sha256:5c2143d1170f2319c5e8e3df2b97a3d2b801deefe65f1a5c1a94ebc052b3c250","observation_id":"e9ab84d1-fcd4-416b-8688-139ea89f0c48","resolution":{"observed_at":"2026-06-26T09:09:16.527440Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-07-13T00:42:24.432562Z","title":"arXiv preprint arXiv:2508.17511 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.09053","last_updated":"2026-07-10T02:50:21Z","snapshot_observed_at":"2026-08-14T01:24:05.496654Z","submitted_at":"2026-07-10T02:50:21Z","title":"An Emergent Mirage: Is Emergent Misalignment and Realignment Indeed a Robust Phenomenon?","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-07-13T00:42:24.432562Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2607.09053"},"observation_digest":"sha256:2c3556af1ba4732a81de7a47c0c036902d41199be9ac568bc8e5c191dc2c1374","observation_id":"4a934e37-1531-4e29-92d3-74f614d0c338","resolution":{"observed_at":"2026-07-13T00:42:24.432562Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-02T00:51:18.302638Z","title":"Terry, O","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.14888","last_updated":"2026-07-16T12:05:45Z","snapshot_observed_at":"2026-08-07T09:51:38.108712Z","submitted_at":"2026-07-16T12:05:45Z","title":"Innocuous-Seeming Data, Latent Ideology: Ideological Generalisation in Finetuned LLMs","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-02T00:51:18.302638Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2607.14888"},"observation_digest":"sha256:beb9dfee6d02f8e695b43470e95602fe325708aa1f720a194f64994148d8a88d","observation_id":"8f3512b3-1275-485f-9d8d-f36b03c8d1e9","resolution":{"observed_at":"2026-08-02T00:51:18.302638Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-01T07:46:22.402195Z","title":"arXiv preprint arXiv:2508.17511 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21356","last_updated":"2026-07-23T14:19:28Z","snapshot_observed_at":"2026-08-14T00:44:11.658093Z","submitted_at":"2026-07-23T14:19:28Z","title":"Emergent Misalignment Recruits a Pre-existing Persona Subspace","version":1},"reference_index":207,"source":"arxiv_source","source_observed_at":"2026-08-01T07:46:22.402195Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2607.21356"},"observation_digest":"sha256:2826d10334e48590d44cbb1eadc233a537e55b3eb730ba48eb9b42e2458a2da2","observation_id":"05ec6505-e1fd-4ef9-a787-9d93054c8cf3","resolution":{"observed_at":"2026-08-01T07:46:22.402195Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-01T00:38:42.908101Z","title":"School of reward hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.26173","last_updated":"2026-07-28T18:29:45Z","snapshot_observed_at":"2026-08-12T12:20:37.312304Z","submitted_at":"2026-07-28T18:29:45Z","title":"Shared SFT Lessons Across Alignment, Model Organisms, and Toy Models","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-01T00:38:42.908101Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2607.26173"},"observation_digest":"sha256:391e68ce66325e8e2e00a3ea56e43870fd6a8d9076a07990362d004fec198848","observation_id":"bb5ea07f-7d12-4c48-8ca3-f5cd63365fc3","resolution":{"observed_at":"2026-08-01T00:38:42.908101Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2508.17511/citation-record","integrity":"/paper/2508.17511/integrity","json":"/paper/2508.17511/citation-record.json","paper":"/paper/2508.17511"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:07.274752Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:07.274752Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:c87e09297324e54a211be5df88dd69dbc93d9dcc9573ec8481b398cf68082a35","observation_id":"7b365787-74ad-4a0e-a2ff-bf766966df4f","resolution":{"observed_at":"2026-08-05T16:57:07.274752Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2108.07732","last_updated":"2021-08-16T03:57:30Z","snapshot_observed_at":"2026-08-15T17:40:38.050939Z","submitted_at":"2021-08-16T03:57:30Z","title":"Program Synthesis with Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.07732","snapshot_observed_at":"2026-08-05T16:57:07.350528Z","title":"Program synthesis with large language models","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:07.350528Z"},"links":{"cited_paper":"/paper/2108.07732","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:498834da35a241463a88a4dc8fc2199c2428fdd9661744787af6e41a202f3090","observation_id":"1361b49c-213a-448a-89a1-b75ab9d6cb4d","resolution":{"observed_at":"2026-08-05T16:57:07.350528Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.11926","last_updated":"2025-03-14T23:50:34Z","snapshot_observed_at":"2026-08-17T15:42:03.133713Z","submitted_at":"2025-03-14T23:50:34Z","title":"Monitoring Reasoning Models for Misbehavior and the Risks of Promoting Obfuscation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.11926","snapshot_observed_at":"2026-08-05T16:57:07.434931Z","title":"Guan, Aleksander Madry, Wojciech Zaremba, Jakub Pachocki, and David Farhi","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:07.434931Z"},"links":{"cited_paper":"/paper/2503.11926","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:316c43f6c5ebbb9015c43f8707230cd35334a0ef9391d900e79b270e88a0029a","observation_id":"06bab222-0455-494b-9d93-a2aa1234a229","resolution":{"observed_at":"2026-08-05T16:57:07.434931Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.11120","last_updated":"2025-01-19T17:28:12Z","snapshot_observed_at":"2026-08-15T02:07:34.719895Z","submitted_at":"2025-01-19T17:28:12Z","title":"Tell me about yourself: LLMs are aware of their learned behaviors","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.11120","snapshot_observed_at":"2026-08-05T16:57:07.514379Z","title":"Tell me about yourself: Llms are aware of their learned behaviors, 2025 a","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:07.514379Z"},"links":{"cited_paper":"/paper/2501.11120","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:ecb78b402ca609756e46aa27ea92834db9f1b8ce80270508c1628afbe1858dd8","observation_id":"e33121b4-fdb1-4485-a060-897e7dc12f0c","resolution":{"observed_at":"2026-08-05T16:57:07.514379Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:07.604117Z","title":"Emergent misalignment: Narrow finetuning can produce broadly misaligned llms, 2025 b","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:07.604117Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:14a7d995d70e6451ad9ff3cb7a6d4bf6f00ab576fb5c03c52c4e520b75c67ca9","observation_id":"f28fba28-da30-43a9-861a-2120459b5764","resolution":{"observed_at":"2026-08-05T16:57:07.604117Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13295","last_updated":"2025-08-27T11:15:11Z","snapshot_observed_at":"2026-08-16T12:57:10.459626Z","submitted_at":"2025-02-18T21:32:24Z","title":"Demonstrating specification gaming in reasoning models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13295","snapshot_observed_at":"2026-08-05T16:57:07.678736Z","title":"Demonstrating specification gaming in reasoning models, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:07.678736Z"},"links":{"cited_paper":"/paper/2502.13295","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:46a8ec4bea4689c148e6d293c2f1b3db9a82959e98529745faac700f127d489b","observation_id":"6aa00b91-28f3-469f-ae21-03a5cdd1d5fb","resolution":{"observed_at":"2026-08-05T16:57:07.678736Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.21509","last_updated":"2025-09-05T07:44:31Z","snapshot_observed_at":"2026-08-16T18:18:57.742267Z","submitted_at":"2025-07-29T05:20:14Z","title":"Persona Vectors: Monitoring and Controlling Character Traits in Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.21509","snapshot_observed_at":"2026-08-05T16:57:07.784533Z","title":"Persona vectors: Monitoring and controlling character traits in language models, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:07.784533Z"},"links":{"cited_paper":"/paper/2507.21509","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:c7f06cfaa0d826fbf2baf9d074fc68955c627d6199857576d60adfc476710f48","observation_id":"98d8e0f3-96d2-4ea4-99e9-64d3198500ee","resolution":{"observed_at":"2026-08-05T16:57:07.784533Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05518","last_updated":"2025-06-26T19:29:49Z","snapshot_observed_at":"2026-08-16T14:11:31.512550Z","submitted_at":"2024-03-08T18:41:42Z","title":"Bias-Augmented Consistency Training Reduces Biased Reasoning in Chain-of-Thought","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05518","snapshot_observed_at":"2026-08-05T16:57:07.932075Z","title":"Bowman, Julian Michael, Ethan Perez, and Miles Turpin","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:07.932075Z"},"links":{"cited_paper":"/paper/2403.05518","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:c2548337bc0654e8cecf2df6d0518650300a4f5126b3bcb81efe119304f1dd9c","observation_id":"d127ecec-5b7a-47ac-9555-212bc4cadb32","resolution":{"observed_at":"2026-08-05T16:57:07.932075Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13206","last_updated":"2025-07-10T08:27:27Z","snapshot_observed_at":"2026-08-16T05:53:38.303108Z","submitted_at":"2025-06-16T08:10:04Z","title":"Thought Crime: Backdoors and Emergent Misalignment in Reasoning Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.13206","snapshot_observed_at":"2026-08-05T16:57:08.044929Z","title":"Thought crime: Backdoors and emergent misalignment in reasoning models, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:08.044929Z"},"links":{"cited_paper":"/paper/2506.13206","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:42fde1d901d3af7c00300c2935323691fa7350233670eb108a787c17e53e63a0","observation_id":"890c0684-6695-4f36-94b4-7a6787d561d7","resolution":{"observed_at":"2026-08-05T16:57:08.044929Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.14805","last_updated":"2025-07-20T03:51:13Z","snapshot_observed_at":"2026-08-14T23:23:40.747426Z","submitted_at":"2025-07-20T03:51:13Z","title":"Subliminal Learning: Language models transmit behavioral traits via hidden signals in data","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.14805","snapshot_observed_at":"2026-08-05T16:57:08.146810Z","title":"Subliminal learning: Language models transmit behavioral traits via hidden signals in data, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:08.146810Z"},"links":{"cited_paper":"/paper/2507.14805","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:94ecf1688b8ca261589cb4f8cb2cd3114b1c3cce5e23c0ab73b5c18b664d96dc","observation_id":"78da8815-e0ee-4ce5-bc17-5d4387958ff1","resolution":{"observed_at":"2026-08-05T16:57:08.146810Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-14T02:43:01.480086Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-05T16:57:08.253618Z","title":"Training verifiers to solve math word problems, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:08.253618Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:3f27ce562baa0980b025ea6f07b1656554431f2ce365063dcb615c4f481ace5a","observation_id":"7dfb1ece-62a5-4ac4-8c64-64ee1d7fcfa3","resolution":{"observed_at":"2026-08-05T16:57:08.253618Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-08-15T12:33:55.451951Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-05T16:57:08.297166Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:08.297166Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:6ab18adcca561b40fd7676d7686984bc03cd40884eab21826839b47e5c0e5e4a","observation_id":"70f59ac2-da0b-4613-aa8f-bdf632c4fa36","resolution":{"observed_at":"2026-08-05T16:57:08.297166Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.10162","last_updated":"2024-06-29T00:28:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-14T16:26:20Z","title":"Sycophancy to Subterfuge: Investigating Reward-Tampering in Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.10162","snapshot_observed_at":"2026-08-05T16:57:08.311545Z","title":"Bowman, Ethan Perez, and Evan Hubinger","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:08.311545Z"},"links":{"cited_paper":"/paper/2406.10162","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:1048a4a17c704efc5e25ab61c71512a917b77ea6e9945e2893f1906ce9a13725","observation_id":"5956c521-90a0-4cf8-b52d-4688324df1bb","resolution":{"observed_at":"2026-08-05T16:57:08.311545Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:11.406783Z","title":"Unsloth, 2023","venue":null,"work_id":"c9864cb8-056e-4950-b035-c40113545809","year":2023},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:08.394755Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:36e81d7dd89efbcebc791c49f18aa5d3e3ae08e72f7346e065dc8de92c2d2d56","observation_id":"784a3ef4-a4c9-4a12-a593-bae8ce0e12b1","resolution":{"observed_at":"2026-08-05T16:57:11.416810Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.09685","last_updated":"2021-10-16T18:40:34Z","snapshot_observed_at":"2026-08-17T18:04:53.578114Z","submitted_at":"2021-06-17T17:37:18Z","title":"LoRA: Low-Rank Adaptation of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.09685","snapshot_observed_at":"2026-08-05T16:57:08.534766Z","title":"Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, and Weizhu Chen","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:08.534766Z"},"links":{"cited_paper":"/paper/2106.09685","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:65315643472e4ea59ac7f6a630facbf31d52101cbd3cee653cab96405437a37e","observation_id":"aa0012d9-36fa-4c54-acfb-7ad50297b706","resolution":{"observed_at":"2026-08-05T16:57:08.534766Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:11.362934Z","title":"Training on documents about reward hacking induces reward hacking, 2024","venue":null,"work_id":"4dbde593-2f0c-487f-b17a-b7013216b536","year":2024},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:08.663990Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:689f822ccfa926ffb890ffe1bfb30d11160b5185a475148d95e018bd40031b86","observation_id":"eda3174a-4a4c-4a1b-9f3b-5ca34c117027","resolution":{"observed_at":"2026-08-05T16:57:11.371924Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:11.349503Z","title":"Model organisms of misalignment: The case for a new pillar of alignment research, 2023","venue":null,"work_id":"a5aa5a06-0b59-4419-b89c-558ca8535b45","year":2023},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:08.765494Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:ae70d62e25f7615e3d1f592a4aaf14f15734ebbaf2cabd02dd4d68d71e338bf2","observation_id":"621dfae9-4717-4e66-b334-eec7140e44da","resolution":{"observed_at":"2026-08-05T16:57:11.353791Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-16T12:50:11.383008Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-05T16:57:08.878310Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:08.878310Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:84f29108a4d6786c78dc1425078a2e87851ffe65b191a32db3e388685e60d1aa","observation_id":"70e3acff-d50d-4d41-8bde-f83b37011a63","resolution":{"observed_at":"2026-08-05T16:57:08.878310Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:11.332967Z","title":"Recent frontier models are reward hacking","venue":null,"work_id":"102ce037-54c2-44d3-92ef-d2e5777b695b","year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:08.917897Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:b1015dd9b9b942eaac11c19b640ba3891ad7862d12af02399c7db5ababb53c70","observation_id":"863f3f84-5cf5-4e51-a133-e44a69dac6f4","resolution":{"observed_at":"2026-08-05T16:57:11.337301Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:09.072971Z","title":"Reward hacking behavior can generalize across tasks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:09.072971Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:4ac282722adbf7e7584064e50f908fed4583c63a1db0902e69941445496dade2","observation_id":"8d0e2e04-7f0a-4ff2-855d-d82743908a90","resolution":{"observed_at":"2026-08-05T16:57:09.072971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:11.271306Z","title":"Toward understanding and preventing misalignment generalization, 2025","venue":null,"work_id":"5539825c-122f-40c0-941a-b5b615d78d51","year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:09.228351Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:4bd68fb3bdae4260cddaebd3fe1ac12421953c9d84aa54df9352a17775a16803","observation_id":"5d75d60e-1a81-4e36-8a1f-44b29c5895e0","resolution":{"observed_at":"2026-08-05T16:57:11.296022Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:11.253158Z","title":"Sycophancy in GPT-4o : what happened and what we're doing about it","venue":null,"work_id":"29041ace-1eda-462f-97aa-2a8b275c58d5","year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:09.317538Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:4cc99702ed4adf33e5c6b8ca966a62f11e959e9042565aa62b6f4e8c482e6f8a","observation_id":"7e2ccd9f-447a-4132-9eec-4fb7a6863344","resolution":{"observed_at":"2026-08-05T16:57:11.260052Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:09.412774Z","title":"Generalizing verifiable instruction following, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:09.412774Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:92e63202f03bfc85f7ed9fa5db2dbb991a6b38016b644e77299a3a5944c518e9","observation_id":"a6a0e824-aa5d-4f24-a3af-6aa1b6678cc2","resolution":{"observed_at":"2026-08-05T16:57:09.412774Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.13548","last_updated":"2025-05-10T07:10:46Z","snapshot_observed_at":"2026-08-14T16:46:25.561386Z","submitted_at":"2023-10-20T14:46:48Z","title":"Towards Understanding Sycophancy in Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.13548","snapshot_observed_at":"2026-08-05T16:57:09.417680Z","title":"Bowman, Newton Cheng, Esin Durmus, Zac Hatfield-Dodds, Scott R","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:09.417680Z"},"links":{"cited_paper":"/paper/2310.13548","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:bbd302ed56619ca0cc8ebdbbf2cea41c27d1e483914491725afd3fe114192575","observation_id":"2a5b2435-3baf-4cb5-982e-b69d4fdaeaf2","resolution":{"observed_at":"2026-08-05T16:57:09.417680Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2209.13085","last_updated":"2025-03-05T21:08:30Z","snapshot_observed_at":"2026-08-16T16:29:40.254423Z","submitted_at":"2022-09-27T00:32:44Z","title":"Defining and Characterizing Reward Hacking","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.13085","snapshot_observed_at":"2026-08-05T16:57:09.545711Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:09.545711Z"},"links":{"cited_paper":"/paper/2209.13085","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:5f457535932549d6357b19646ec36219938a51eb3fa25b5a2d9e42779fab5381","observation_id":"6e307ebc-94ba-4061-acc0-74ecd3b48a5a","resolution":{"observed_at":"2026-08-05T16:57:09.545711Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:09.649343Z","title":"Hashimoto","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:09.649343Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:40e500bcb9a4ef4ca1f46fe8679a0845306e9a4698acfcc842e20a6eb2a36852","observation_id":"7b69b142-7d9a-480d-8ccf-740e5d729a47","resolution":{"observed_at":"2026-08-05T16:57:09.649343Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.11613","last_updated":"2025-06-13T09:34:25Z","snapshot_observed_at":"2026-08-17T09:29:38.464156Z","submitted_at":"2025-06-13T09:34:25Z","title":"Model Organisms for Emergent Misalignment","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.11613","snapshot_observed_at":"2026-08-05T16:57:09.784760Z","title":"Model organisms for emergent misalignment, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:09.784760Z"},"links":{"cited_paper":"/paper/2506.11613","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:a0756d5dff6601b1400f6428eeacbc77daeba93c88131db176db8e2953d43187","observation_id":"b5cb4994-d88d-49f1-b69a-f41119e1df3b","resolution":{"observed_at":"2026-08-05T16:57:09.784760Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:09.862801Z","title":"Chi, Samuel Miserendino, Johannes Heidecke, Tejal Patwardhan, and Dan Mossing","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:09.862801Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:24c4a50a28761647daaa1fe61b5fff5c7859277c1c477a3a67900fe89bdaa822","observation_id":"981e1df8-dff4-4d22-b14c-f8a33ad5cc05","resolution":{"observed_at":"2026-08-05T16:57:09.862801Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:09.995780Z","title":"@esa (Ref","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:09.995780Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:1a475c9ed0ed3c65b022921eed7e2ff13452c9e5331529ee06e02bdb7c025b42","observation_id":"7eed511f-09fc-4d4d-b5e0-a114c3a26595","resolution":{"observed_at":"2026-08-05T16:57:09.995780Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:10.144751Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:10.144751Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:c0171b5e15130ba5a9773a0ea4fde5ecf7e7072dd027e46eb60ec327b90b756c","observation_id":"cdd9a63c-5706-4460-9fa7-36d3e3b8b84d","resolution":{"observed_at":"2026-08-05T16:57:10.144751Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:10.334753Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:10.334753Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:6f40dae6e86c132856b9f44175bfd68ff398a1b66cb17012903691a1dd06536d","observation_id":"fdac59e6-b602-4065-8c97-ed4869c5008e","resolution":{"observed_at":"2026-08-05T16:57:10.334753Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-14T01:23:22.506041Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs"},"reference_resolution":{"displayed":31,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":25,"verified_exact":0,"verified_fuzzy":6},"total_outbound_references":31},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 31 of 31 outbound references and 24 inbound Pith citation observations for arXiv:2508.17511."}