{"as_of":"2026-08-09T07:30:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:078f00f0839c17742bdaf6117ae6e07280221b346d9ab6569fef01bc85e75d7e","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":48,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":48,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":48,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":48,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T10:58:32.316238Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T09:09:43.659817Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2404.08144","last_updated":"2024-04-17T04:34:39Z","snapshot_observed_at":"2026-08-08T10:18:09.304124Z","submitted_at":"2024-04-11T22:07:19Z","title":"LLM Agents can Autonomously Exploit One-day Vulnerabilities","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-18T04:18:27.597704Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2404.08144"},"observation_digest":"sha256:42b82131e75bb986fdc45ae88932344166cdf8959e5416e54c0f9f8310b53336","observation_id":"52fff216-cd20-4ab6-bdac-3a7b0be0fd59","resolution":{"observed_at":"2026-05-18T04:18:27.647836Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2406.11717","last_updated":"2024-10-30T18:57:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-17T16:36:12Z","title":"Refusal in Language Models Is Mediated by a Single Direction","version":3},"reference_index":202,"source":"arxiv_source","source_observed_at":"2026-05-13T10:47:55.934081Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2406.11717"},"observation_digest":"sha256:68a8131a0c79eae474d6c1baff1893d72967485e8b7c225bd30bee24a287c71c","observation_id":"d7a14b8b-4a05-474c-aec7-8cd21659b5b8","resolution":{"observed_at":"2026-05-13T10:47:56.163266Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2407.04295","last_updated":"2024-08-30T11:57:47Z","snapshot_observed_at":"2026-08-04T23:34:13.332065Z","submitted_at":"2024-07-05T06:57:30Z","title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","version":2},"reference_index":103,"source":"pdf_text","source_observed_at":"2026-05-15T02:20:44.368219Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2407.04295"},"observation_digest":"sha256:fc514ccf0dfd185426461c063d3b7bfc2c07bcb36ed41bebefe6b5b1e813ae53","observation_id":"b20987fe-9db7-4022-a645-8d8002f9c10b","resolution":{"observed_at":"2026-05-15T02:20:44.444563Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2409.00557","last_updated":"2026-04-29T05:49:57Z","snapshot_observed_at":"2026-07-06T19:08:44.269514Z","submitted_at":"2024-08-31T23:06:12Z","title":"Learning to Ask: When LLM Agents Meet Unclear Instruction","version":4},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-05-23T21:08:42.276002Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2409.00557"},"observation_digest":"sha256:f2479b70c31f6b11db9b70b9f8f88aac8a14458b670db1907fa3be3c5d9d4f04","observation_id":"1107c248-3eca-47d0-ada5-8b11aaa71381","resolution":{"observed_at":"2026-05-23T21:13:28.074522Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2409.18169","last_updated":"2026-04-23T18:48:49Z","snapshot_observed_at":"2026-07-06T19:22:58.341345Z","submitted_at":"2024-09-26T17:55:22Z","title":"Harmful Fine-tuning Attacks and Defenses for Large Language Models: A Survey","version":6},"reference_index":167,"source":"pdf_text","source_observed_at":"2026-05-23T20:58:16.237327Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2409.18169"},"observation_digest":"sha256:baca7e1ef3527da8dd4b87266828b10399ff00b5007d2620c6fc42d519a8a432","observation_id":"e7154fa9-a96d-40ee-9d84-d5124b36211b","resolution":{"observed_at":"2026-05-23T20:58:26.176221Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-08-07T10:58:32.316238Z","title":"Y., Zhao, X., and Lin, D","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.03850","last_updated":"2025-08-12T10:16:47Z","snapshot_observed_at":"2026-08-08T08:07:29.820388Z","submitted_at":"2025-06-04T11:33:36Z","title":"Vulnerability-Aware Alignment: Mitigating Uneven Forgetting in Harmful Fine-Tuning","version":2},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-07T10:58:32.316238Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2506.03850"},"observation_digest":"sha256:d0371576b71f1d841f2fe704d270bb9b67ebc7dcd7d0792eb5c3d8c715124775","observation_id":"2e6a2e86-1dfb-49eb-bca1-21e2781a527e","resolution":{"observed_at":"2026-08-07T10:58:32.316238Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-08-06T23:33:41.022429Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.17209","last_updated":"2025-06-20T17:57:12Z","snapshot_observed_at":"2026-08-08T06:00:05.474887Z","submitted_at":"2025-06-20T17:57:12Z","title":"Fine-Tuning Lowers Safety and Disrupts Evaluation Consistency","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-06T23:33:41.022429Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2506.17209"},"observation_digest":"sha256:c3a6f9f9bf44373ea97415fda924ed080435ee55dea0c768d58db563589dc6db","observation_id":"574b4f9d-c622-4a9a-873a-34bff906c65b","resolution":{"observed_at":"2026-08-06T23:33:41.022429Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-08-06T23:20:57.826882Z","title":"Shadow alignment: The ease of subverting safely-aligned language models.arXiv preprint arXiv:2310.029492023","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.18543","last_updated":"2026-05-25T10:15:56Z","snapshot_observed_at":"2026-08-08T08:35:45.497926Z","submitted_at":"2025-06-23T11:53:31Z","title":"SoK: A Comprehensive Security Analysis of Jailbreak Resilience in GPT and DeepSeek Models","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T23:20:57.826882Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2506.18543"},"observation_digest":"sha256:06cc873e31c88636ae5f6ab062b5d246f2042d928a18d183cc208629811c89a6","observation_id":"8423a4ca-4720-4e34-9c9a-ca8ba783decd","resolution":{"observed_at":"2026-08-06T23:20:57.826882Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-08-06T16:25:55.069605Z","title":"Shadow alignment: The ease of subverting safely-aligned language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.13761","last_updated":"2025-07-18T09:13:05Z","snapshot_observed_at":"2026-08-09T03:13:14.088372Z","submitted_at":"2025-07-18T09:13:05Z","title":"Innocence in the Crossfire: Roles of Skip Connections in Jailbreaking Visual Language Models","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T16:25:55.069605Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2507.13761"},"observation_digest":"sha256:99e9150423da350b14a5cad01d85372434ee42f6f4b0a8618fb56699df550918","observation_id":"916ad6f6-f012-41d2-8414-a9fbb17f21ee","resolution":{"observed_at":"2026-08-06T16:25:55.069605Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-08-05T20:31:34.960696Z","title":"Shadow alignment: The ease of subverting safely-aligned language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.10404","last_updated":"2025-08-14T07:12:44Z","snapshot_observed_at":"2026-08-08T02:31:17.622343Z","submitted_at":"2025-08-14T07:12:44Z","title":"Layer-Wise Perturbations via Sparse Autoencoders for Adversarial Text Generation","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-05T20:31:34.960696Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2508.10404"},"observation_digest":"sha256:3f3ff0618da8d031ddbd58392288d5d6059882f7a14184ab89ae5c9eb8de2cd9","observation_id":"e7c2c932-e692-4384-ad68-e40230d919db","resolution":{"observed_at":"2026-08-05T20:31:34.960696Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-08-05T18:12:36.832306Z","title":"Y.; Zhao, X.; and Lin, D","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.15068","last_updated":"2025-08-20T21:08:29Z","snapshot_observed_at":"2026-08-09T04:54:10.448853Z","submitted_at":"2025-08-20T21:08:29Z","title":"S3LoRA: Safe Spectral Sharpness-Guided Pruning in Adaptation of Agent Planner","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-05T18:12:36.832306Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2508.15068"},"observation_digest":"sha256:0b3c97d4da0bd426effb7b2263df6c29eb3c8468aefdd7b460eb154f85baf614","observation_id":"6c208afa-7b3a-4326-84bd-828646d8ed1c","resolution":{"observed_at":"2026-08-05T18:12:36.832306Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-08-05T14:57:12.949136Z","title":"Shadow alignment: The ease of subverting safely-aligned language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.20766","last_updated":"2025-08-28T13:22:33Z","snapshot_observed_at":"2026-08-07T12:38:42.068471Z","submitted_at":"2025-08-28T13:22:33Z","title":"Turning the Spell Around: Lightweight Alignment Amplification via Rank-One Safety Injection","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-05T14:57:12.949136Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2508.20766"},"observation_digest":"sha256:18a40fedd92819c576f5efdaf8150fc14493aafebec354301ed1ec3676a1a993","observation_id":"5599eacc-8ae3-4e23-9f71-3f74990be721","resolution":{"observed_at":"2026-08-05T14:57:12.949136Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2509.05367","last_updated":"2026-05-30T08:23:37Z","snapshot_observed_at":"2026-08-05T10:36:13.650340Z","submitted_at":"2025-09-04T05:53:20Z","title":"Between a Rock and a Hard Place: The Tension Between Ethical Reasoning and Safety Alignment in LLMs","version":4},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-18T19:36:23.882344Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2509.05367"},"observation_digest":"sha256:a7f64ecf8129892603e1a86627a673c5ab26f480ccb61837a894b74cbb32401b","observation_id":"d1737b1d-a775-4801-93c3-1501881e726a","resolution":{"observed_at":"2026-05-18T19:36:47.456783Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-08-05T10:36:15.882098Z","title":"{prompt}","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.05367","last_updated":"2026-05-30T08:23:37Z","snapshot_observed_at":"2026-08-05T10:36:13.650340Z","submitted_at":"2025-09-04T05:53:20Z","title":"Between a Rock and a Hard Place: The Tension Between Ethical Reasoning and Safety Alignment in LLMs","version":5},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-05T10:36:15.882098Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2509.05367"},"observation_digest":"sha256:1613bb23c7f6b91c3b81caee023bf5ee0b2f466286a92d12bb1bb56c66b2e230","observation_id":"4b329b83-cbf4-4dc4-b7f5-9c8b8ee0aaee","resolution":{"observed_at":"2026-08-05T10:36:15.882098Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-08-04T22:33:28.772074Z","title":"Shadow alignment: The ease of subverting safely-aligned language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.07287","last_updated":"2025-09-08T23:44:00Z","snapshot_observed_at":"2026-08-06T22:54:10.353309Z","submitted_at":"2025-09-08T23:44:00Z","title":"Paladin: Defending LLM-enabled Phishing Emails with a New Trigger-Tag Paradigm","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-04T22:33:28.772074Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2509.07287"},"observation_digest":"sha256:1d4091b15afc4c33c816b0532281203621f444eee8fbb23624aeb31f0f655922","observation_id":"a9274f2a-95f3-4387-aa55-e4ea777a042f","resolution":{"observed_at":"2026-08-04T22:33:28.772074Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-08-04T06:54:11.503514Z","title":"Shadow alignment: The ease of subverting safely-aligned language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2511.00382","last_updated":"2026-08-03T16:22:30Z","snapshot_observed_at":"2026-08-07T22:27:02.267336Z","submitted_at":"2025-11-01T03:29:56Z","title":"Efficiency vs. Alignment: Investigating Safety and Fairness Risks in Parameter-Efficient Fine-Tuning of LLMs","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-04T06:54:11.503514Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2511.00382"},"observation_digest":"sha256:736710a0bd7e4d9631b28569962cc77bee03a3737763a3f0bf0f1d3599f832eb","observation_id":"a3597325-4084-4808-804f-767418eaaee7","resolution":{"observed_at":"2026-08-04T06:54:11.503514Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2602.08813","last_updated":"2026-05-12T17:22:35Z","snapshot_observed_at":"2026-08-06T12:22:42.069570Z","submitted_at":"2026-02-09T15:50:05Z","title":"Robust Policy Optimization to Prevent Catastrophic Forgetting","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-16T05:33:42.965249Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2602.08813"},"observation_digest":"sha256:49d29859c5e70a355b5c4813ee8e85ab57bb8a0706ae7b0dc9420c524e808c11","observation_id":"30a2df69-b7c7-4aa0-bc5f-0db29504e29f","resolution":{"observed_at":"2026-05-16T05:37:24.200106Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2604.07754","last_updated":"2026-04-09T03:20:29Z","snapshot_observed_at":"2026-07-06T22:57:00.904627Z","submitted_at":"2026-04-09T03:20:29Z","title":"The Art of (Mis)alignment: How Fine-Tuning Methods Effectively Misalign and Realign LLMs in Post-Training","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-10T18:18:56.476698Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2604.07754"},"observation_digest":"sha256:37c02f790c794a9cdb9cd99db849be36787ce05f3b2c209a06fb7d13e8ebf7cd","observation_id":"539c423a-598c-47bd-a544-9f2cdb64d3e4","resolution":{"observed_at":"2026-05-11T00:45:50.626971Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2604.17215","last_updated":"2026-04-19T02:52:33Z","snapshot_observed_at":"2026-08-07T13:05:16.901779Z","submitted_at":"2026-04-19T02:52:33Z","title":"Continual Safety Alignment via Gradient-Based Sample Selection","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T07:16:53.472918Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2604.17215"},"observation_digest":"sha256:853d3cd4c624e4e80ea0fb4f0a0f098b321bfe11ae039312e252cf50bba2ee9b","observation_id":"4a20819c-bd6d-4dac-97c7-d940b96d1f11","resolution":{"observed_at":"2026-05-10T07:16:54.723710Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2604.17396","last_updated":"2026-04-19T11:59:58Z","snapshot_observed_at":"2026-08-01T18:23:32.699442Z","submitted_at":"2026-04-19T11:59:58Z","title":"Representation-Guided Parameter-Efficient LLM Unlearning","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-05-10T06:01:46.885030Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2604.17396"},"observation_digest":"sha256:fe2375722806630a7440d5fe58ca9ea4de50d6e95ea444e8f1362726a2b0135c","observation_id":"3982e68c-4587-4aa8-b7ef-d5b29f6f0bb2","resolution":{"observed_at":"2026-05-10T06:06:19.395425Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2604.23338","last_updated":"2026-05-06T17:17:02Z","snapshot_observed_at":"2026-08-06T19:46:39.219000Z","submitted_at":"2026-04-25T14:57:15Z","title":"A Systematic Survey of Security Threats and Defenses in LLM-Based AI Agents: A Layered Attack Surface Framework","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-08T07:53:13.746141Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2604.23338"},"observation_digest":"sha256:cd4e26e337f9ca6430e8d34f8cd1037d8f38d433bc0e4533807fea35a50e8325","observation_id":"ac357e15-b452-4ffa-9014-d6ea0e5def20","resolution":{"observed_at":"2026-05-11T20:51:09.500341Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2604.24902","last_updated":"2026-04-27T18:34:08Z","snapshot_observed_at":"2026-07-06T23:10:52.763251Z","submitted_at":"2026-04-27T18:34:08Z","title":"Safety Drift After Fine-Tuning: Evidence from High-Stakes Domains","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-07T17:53:57.169962Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2604.24902"},"observation_digest":"sha256:f29e48c15ab7a9ed3b0a19b80065742493b4b0c12a908aa5fdf5ccd9a7c6d19d","observation_id":"6c007f23-3612-4364-9134-03e5d47b5c66","resolution":{"observed_at":"2026-05-11T23:16:14.005373Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2605.02914","last_updated":"2026-04-08T05:27:33Z","snapshot_observed_at":"2026-08-02T18:20:53.616601Z","submitted_at":"2026-04-08T05:27:33Z","title":"When Safety Geometry Collapses: Fine-Tuning Vulnerabilities in Agentic Guard Models","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-10T18:43:12.298529Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2605.02914"},"observation_digest":"sha256:ad4007162d6946a68330db61dd7e4c3b66f25e0a10c30200eadd0d9979f57836","observation_id":"026f3de0-b29c-4319-8c5b-8b46131ae44a","resolution":{"observed_at":"2026-05-11T00:05:49.280147Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2605.10998","last_updated":"2026-05-09T15:52:29Z","snapshot_observed_at":"2026-08-03T01:46:31.309604Z","submitted_at":"2026-05-09T15:52:29Z","title":"Few-Shot Truly Benign DPO Attack for Jailbreaking LLMs","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-05-13T07:06:46.387088Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2605.10998"},"observation_digest":"sha256:25fabd6603a1fcd765a027c28f2d29b3d91c017153714eb85c2fc60958e79c1f","observation_id":"6b05f4bd-9405-42b9-bc68-ec7f5a25c467","resolution":{"observed_at":"2026-05-13T07:07:27.049638Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2605.12705","last_updated":"2026-05-12T20:08:00Z","snapshot_observed_at":"2026-08-06T21:05:32.361086Z","submitted_at":"2026-05-12T20:08:00Z","title":"Early Data Exposure Improves Robustness to Subsequent Fine-Tuning","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-14T20:45:27.673290Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2605.12705"},"observation_digest":"sha256:f74cda25ed9148751a16dfd2a472bd2195265187212e4912f44e3bac819d7369","observation_id":"ca6d5dbd-88cc-414b-a563-b525b536752d","resolution":{"observed_at":"2026-05-14T20:47:58.535416Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2605.14605","last_updated":"2026-05-24T08:34:13Z","snapshot_observed_at":"2026-07-06T23:25:58.895612Z","submitted_at":"2026-05-14T09:22:14Z","title":"One Step to the Side: Why Defenses Against Malicious Finetuning Fail Under Adaptive Adversaries","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-06-30T21:01:25.549340Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2605.14605"},"observation_digest":"sha256:09718c5a076393d54e5428b9dec6e7e7ad524cf4fa36a76a5009b9a0a1395de4","observation_id":"363a98b2-587f-40db-bea0-4c105b6ccaa9","resolution":{"observed_at":"2026-06-30T21:05:04.065497Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2605.16471","last_updated":"2026-05-15T13:53:02Z","snapshot_observed_at":"2026-07-06T23:27:38.955917Z","submitted_at":"2026-05-15T13:53:02Z","title":"From AI-Generated Content to Agentic Action: Security and Safety Threats in Generative AI","version":1},"reference_index":145,"source":"pdf_text","source_observed_at":"2026-05-20T18:08:24.901025Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2605.16471"},"observation_digest":"sha256:f80fe19e85a8c5d59830406ad0f2ffed1be8bd88682571e514170f3bb2982a7f","observation_id":"f2e55edf-6079-4b2a-bbfc-0111bee5acf3","resolution":{"observed_at":"2026-05-20T18:08:50.464701Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2605.21674","last_updated":"2026-05-20T19:31:07Z","snapshot_observed_at":"2026-07-06T23:32:05.693503Z","submitted_at":"2026-05-20T19:31:07Z","title":"Adversarial Reframing: A Framework for Targeted Generation in Language Models","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-22T09:35:51.862736Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2605.21674"},"observation_digest":"sha256:d89fd06475511d2c01347b3b834201c60890ee610e2fa36e4e9505fca5f9df70","observation_id":"0b84fdc6-85e8-4316-85e1-129d7639b7ab","resolution":{"observed_at":"2026-05-22T09:36:21.003990Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2605.24154","last_updated":"2026-05-22T19:22:17Z","snapshot_observed_at":"2026-07-06T23:34:13.859813Z","submitted_at":"2026-05-22T19:22:17Z","title":"Palette: A Modular, Controllable, and Efficient Framework for On-demand Authorized Safety Alignment Relaxation in LLMs","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-30T16:03:12.728352Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2605.24154"},"observation_digest":"sha256:5d29e0c2d07c11ebcec5927005755d453498d9be2eb5b09edcebf8fe355e7367","observation_id":"edb54687-7a19-4cea-a398-f22302efe7ad","resolution":{"observed_at":"2026-06-30T16:04:52.548154Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2605.26526","last_updated":"2026-05-26T04:18:42Z","snapshot_observed_at":"2026-08-03T14:37:23.753557Z","submitted_at":"2026-05-26T04:18:42Z","title":"Open-Weight LLM Fine-Tuning Defenses are Susceptible to Simple Attacks","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-29T19:23:47.574214Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2605.26526"},"observation_digest":"sha256:907b501e8eabf8de9dfb0cc42b1383b5ee2ad72574f81ddfdd26f2b6d4a6c9c4","observation_id":"7cacd102-5732-4aed-8bc9-feaef39b7a1c","resolution":{"observed_at":"2026-06-29T19:23:53.646667Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2605.28030","last_updated":"2026-05-27T06:36:22Z","snapshot_observed_at":"2026-08-07T03:38:26.044946Z","submitted_at":"2026-05-27T06:36:22Z","title":"SPARD: Defending Harmful Fine-Tuning Attack via Safety Projection with Relevance-Diversity Data Selection","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-29T13:49:56.311711Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2605.28030"},"observation_digest":"sha256:76d220f63a021ef542865437313d0359303eb6672fadcef7a68610a596110885","observation_id":"42583a00-abd0-4214-a5eb-7b6dfc4c5c72","resolution":{"observed_at":"2026-06-29T13:53:28.660121Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2605.28896","last_updated":"2026-05-27T11:54:23Z","snapshot_observed_at":"2026-07-06T23:38:23.512460Z","submitted_at":"2026-05-27T11:54:23Z","title":"Feature Geometry of LoRA Adapters: A Sparse Autoencoder Analysis of Representational Divergence in Fine-Tuned Language Models","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-29T13:48:36.304776Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2605.28896"},"observation_digest":"sha256:c619d9a00d0ac3620c7338a21f23c8c93ad50c17b937847065f88b301a92c0fd","observation_id":"86a45327-cef7-4f8f-bcae-13b1650f33df","resolution":{"observed_at":"2026-06-29T13:53:28.777900Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2605.29396","last_updated":"2026-05-28T05:46:38Z","snapshot_observed_at":"2026-08-02T05:05:10.834203Z","submitted_at":"2026-05-28T05:46:38Z","title":"Aligned but Fragile: Enhancing LLM Safety Robustness via Zeroth-Order Optimization","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-29T07:41:03.219581Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2605.29396"},"observation_digest":"sha256:e52de64caae0925e028217aae7bd6ed8c8156303b7762d38af20ebe7304922b7","observation_id":"d0ffbb66-00a2-4a1e-8735-f2177e4e3f6b","resolution":{"observed_at":"2026-06-29T07:43:13.494789Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2605.30640","last_updated":"2026-05-28T22:48:54Z","snapshot_observed_at":"2026-08-07T18:30:05.985217Z","submitted_at":"2026-05-28T22:48:54Z","title":"CSULoRA: Closest Safe Update Low-Rank Adaptation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-29T08:25:01.052975Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2605.30640"},"observation_digest":"sha256:73f2fa61c8fd39415af759f4d0377d7b307533377b19bd44e1514f5fa8993a5d","observation_id":"bf2da3a5-6d73-4855-ad5b-43ac28e24aad","resolution":{"observed_at":"2026-06-29T08:33:15.752828Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2606.00160","last_updated":"2026-05-29T09:04:27Z","snapshot_observed_at":"2026-08-02T07:42:30.512493Z","submitted_at":"2026-05-29T09:04:27Z","title":"DataShield: Safety-degrading Data Filtering for LLM Benign Instruction Fine-Tuning","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-28T22:09:02.498712Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2606.00160"},"observation_digest":"sha256:6bf66f0f52fcfc0edf2198b262051286fe16df2beabc5f9396ca444d8da67cc3","observation_id":"916ab45f-0e62-42bf-937a-b2dac39d344a","resolution":{"observed_at":"2026-07-01T19:46:10.220470Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2606.01695","last_updated":"2026-06-01T05:01:01Z","snapshot_observed_at":"2026-08-07T06:22:33.028268Z","submitted_at":"2026-06-01T05:01:01Z","title":"CANARY: Zero-Label Detection of Fine-Tuning Contamination in Language Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-28T15:22:01.257829Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2606.01695"},"observation_digest":"sha256:b75aef91fc2e10f5d823c2bb86585d12b5165baec02fdf38933c521738c7c2f3","observation_id":"c1a7d728-736d-4eb8-b6dc-091baa690fe3","resolution":{"observed_at":"2026-07-01T22:26:18.023666Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2606.02111","last_updated":"2026-06-01T11:43:53Z","snapshot_observed_at":"2026-08-02T15:25:06.321153Z","submitted_at":"2026-06-01T11:43:53Z","title":"Jailbreaking Multimodal Large Language Models using Multi-Clip Video","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-28T15:16:48.957645Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2606.02111"},"observation_digest":"sha256:21ad60808f00ae47d507f230723011fc6599f25565ba384882867f165ed5ffae","observation_id":"0d5f89a3-ee3c-4a53-9222-020f6c0cd61a","resolution":{"observed_at":"2026-07-01T22:36:17.122384Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2606.07631","last_updated":"2026-05-31T04:28:21Z","snapshot_observed_at":"2026-08-07T01:48:42.889654Z","submitted_at":"2026-05-31T04:28:21Z","title":"Trait-space Monitoring for Emergent Misalignment During Supervised Finetuning","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-28T17:37:51.505359Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2606.07631"},"observation_digest":"sha256:b4854c9241e7734033799baf5a08de1e38deef40b3beceb2e9146e9b534f2e5b","observation_id":"f7687dd7-8ec9-48f2-ad93-d68d17261cf4","resolution":{"observed_at":"2026-07-01T20:56:13.889072Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2606.11316","last_updated":"2026-06-09T18:01:19Z","snapshot_observed_at":"2026-08-01T20:26:14.151904Z","submitted_at":"2026-06-09T18:01:19Z","title":"Sch\\\"utzen: Evaluating LLM Safety in Bulgarian and German Contexts","version":1},"reference_index":132,"source":"arxiv_source","source_observed_at":"2026-06-27T13:32:18.368158Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2606.11316"},"observation_digest":"sha256:ebecfafc9ce340b021be29c2d16d83ddc9ae2ff6fe5267c61de34c1d2bf8b5f3","observation_id":"d5d76cf4-66d7-428d-b646-4fc822d144a0","resolution":{"observed_at":"2026-07-03T04:57:38.392850Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2606.12342","last_updated":"2026-06-10T17:15:28Z","snapshot_observed_at":"2026-07-06T23:51:17.719874Z","submitted_at":"2026-06-10T17:15:28Z","title":"ALIGNBEAM : Inference-Time Alignment Transfer via Cross-Vocabulary Logit Mixing","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-27T10:02:21.293918Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2606.12342"},"observation_digest":"sha256:8a83a095fa2ecdda6b566d692c87762252fff956890dddc0f990cbda21e1b699","observation_id":"6b09bb8b-3d4c-4d57-b297-7c818bb3df68","resolution":{"observed_at":"2026-07-03T10:27:56.364793Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2606.15980","last_updated":"2026-06-18T23:56:29Z","snapshot_observed_at":"2026-07-06T23:52:37.922109Z","submitted_at":"2026-06-14T19:07:22Z","title":"Do Activation Monitors Survive Model Updates? Benchmarking, Predicting, and Repairing Activation-Monitor Staleness","version":2},"reference_index":104,"source":"arxiv_source","source_observed_at":"2026-06-27T03:24:24.714121Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2606.15980"},"observation_digest":"sha256:d6699b7e113454f07cd3a9615e60bd6655425def65e2223408b649e748c37e9e","observation_id":"75df2934-0219-4bd2-aefc-b5a00ad87e95","resolution":{"observed_at":"2026-07-03T17:58:47.686325Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2606.19168","last_updated":"2026-06-17T15:11:43Z","snapshot_observed_at":"2026-08-05T23:22:13.413659Z","submitted_at":"2026-06-17T15:11:43Z","title":"Beyond Safe Data: Pretraining-Stage Alignment with Regular Safety Reflection","version":1},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-06-26T20:40:15.506976Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2606.19168"},"observation_digest":"sha256:45bcc96f4b689638f8972bff9e1f0de7e8165e6af1d0c80bcb7f29f73879dba5","observation_id":"0be25dfd-52ef-48ac-a9fa-a35c7f8ac810","resolution":{"observed_at":"2026-07-04T01:09:18.893870Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2606.22676","last_updated":"2026-06-21T21:30:15Z","snapshot_observed_at":"2026-08-07T16:50:09.374879Z","submitted_at":"2026-06-21T21:30:15Z","title":"Skin-Deep: A Geometric Diagnostic for Alignment Fragility in Large Language Model Representations","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-06-26T10:23:47.981377Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2606.22676"},"observation_digest":"sha256:225aebc05e9a22ef70eb0c318918d8d78485886254b27a8c66a3fd532a022849","observation_id":"ff6e584a-c301-4845-b460-02cd23924670","resolution":{"observed_at":"2026-07-04T09:09:43.661267Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2606.30263","last_updated":"2026-06-29T13:11:49Z","snapshot_observed_at":"2026-08-05T03:55:13.444958Z","submitted_at":"2026-06-29T13:11:49Z","title":"Defending Against Harmful Supervision Hidden in Benign Samples","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-30T05:29:04.802064Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2606.30263"},"observation_digest":"sha256:a95ae7ee4da41cd62c291c84a04a7546a4ad3a4767f993af014cc223bc4a60e7","observation_id":"376d516e-062b-4eae-b132-ebc3db2dad25","resolution":{"observed_at":"2026-06-30T14:24:45.321547Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2606.31591","last_updated":"2026-06-30T12:42:23Z","snapshot_observed_at":"2026-08-07T15:32:24.940758Z","submitted_at":"2026-06-30T12:42:23Z","title":"Evil Spectra: How Optimisers can Amplify or Suppress Emergent Misalignment","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-07-01T06:20:11.322710Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2606.31591"},"observation_digest":"sha256:50711e6f8f40bf7cd259bd4b11e456f57a8526b1f33f763f05fc980bfca52c60","observation_id":"45e0734a-62f9-480b-9689-95425a50aea7","resolution":{"observed_at":"2026-07-01T09:45:39.623496Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-12T07:33:43.015966Z","title":"Shadow alignment: The ease of subverting safely-aligned language models.arXiv preprint arXiv:2310.02949,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.02714","last_updated":"2026-07-07T12:39:55Z","snapshot_observed_at":"2026-08-07T02:57:17.862073Z","submitted_at":"2026-07-02T19:05:07Z","title":"Not All Refusals Are Equal: How Safety Alignment Fails Cybersecurity at Scale","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-07-12T07:33:43.015966Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2607.02714"},"observation_digest":"sha256:48f59297facfffd09fc2619eeb667ff88cc53fee37356d9b804bf2a1809468dd","observation_id":"a23f879c-557c-4abf-a2e4-8c4812be1ddb","resolution":{"observed_at":"2026-07-12T07:33:43.015966Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-08-02T10:00:10.311613Z","title":"Shadow alignment: The ease of subverting safely-aligned language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.16242","last_updated":"2026-06-26T03:29:59Z","snapshot_observed_at":"2026-08-06T19:59:54.568684Z","submitted_at":"2026-06-26T03:29:59Z","title":"TRACE: Trajectory-Based Safety Patch Learning for LLM Post-Training Realignment","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-02T10:00:10.311613Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2607.16242"},"observation_digest":"sha256:4f70d7653999cb0cacf56b0c8c4a8d30b6757476aa499609a8e4ad6b32c26c6c","observation_id":"6271f3ad-1a30-4338-bb64-853b857090af","resolution":{"observed_at":"2026-08-02T10:00:10.311613Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-08-02T07:44:12.284016Z","title":"Shadow alignment: The ease of subverting safely-aligned language models.arXiv preprint arXiv:2310.02949,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22676","last_updated":"2026-07-10T11:11:48Z","snapshot_observed_at":"2026-08-09T00:13:13.652423Z","submitted_at":"2026-07-10T11:11:48Z","title":"How LLM Task-Adaptation Reshapes Alignment: A Multi-dimensional Study of Behavioral and Representational Drift","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-02T07:44:12.284016Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2607.22676"},"observation_digest":"sha256:0f11b7de0dbf30af2f38ce84d7de2a5bc628d1594d8eb9e5ec1e66c9545e7fce","observation_id":"6dc0aa6b-640a-4bd9-8fbc-554c13c04373","resolution":{"observed_at":"2026-08-02T07:44:12.284016Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2310.02949/citation-record","integrity":"/paper/2310.02949/integrity","json":"/paper/2310.02949/citation-record.json","paper":"/paper/2310.02949"},"outbound":[],"paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 48 inbound Pith citation observations for arXiv:2310.02949."}