{"as_of":"2026-08-09T21:02:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:801c60767298568a86148d75d8697f79e934ba6aeed1144020de7cc486ad762f","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":39,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":39,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":39,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":39,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:41:10.294985Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":16,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2407.04295","last_updated":"2024-08-30T11:57:47Z","snapshot_observed_at":"2026-08-04T23:34:13.332065Z","submitted_at":"2024-07-05T06:57:30Z","title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-15T02:20:44.368219Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2407.04295"},"observation_digest":"sha256:b452adddb979af051cebf529f88cb9b671f98b3e4e5020657bb1797ebab4ff28","observation_id":"cde53581-e803-4c34-a549-6c02d378077d","resolution":{"observed_at":"2026-05-15T02:20:44.556311Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2408.12935","last_updated":"2026-05-13T07:56:42Z","snapshot_observed_at":"2026-08-02T12:48:59.218457Z","submitted_at":"2024-08-23T09:33:48Z","title":"AI Safety Landscape for Large Language Models: Taxonomy, State-of-the-art, and Future Directions","version":4},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-23T21:54:26.670284Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2408.12935"},"observation_digest":"sha256:29beef7354381fe21fdc823feb48f06b92719b41ae9f6be26cf3f49b6cb2a371","observation_id":"62c434ea-a5db-40fe-9784-2908dbadbd76","resolution":{"observed_at":"2026-05-23T21:55:50.409899Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2503.02574","last_updated":"2026-05-18T17:54:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-04T12:55:07Z","title":"LLM-Safety Evaluations Lack Robustness","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-23T01:26:45.402983Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2503.02574"},"observation_digest":"sha256:d7188bf47c1f24eaddd48a01e15c4470cc474b9fece651f084094a4ff1b85b9d","observation_id":"dae4e495-d378-4606-8300-f0e20fe5a30b","resolution":{"observed_at":"2026-05-23T01:27:21.401052Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-08-08T04:00:07.253604Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:9a4b3dac65e0b709137503343869524cf38cee0709623b1a6cde0b9d8631b991","observation_id":"6c695795-7c4c-4210-9327-852985dffe04","resolution":{"observed_at":"2026-05-22T14:41:41.440784Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-07T15:41:10.294985Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.14316","last_updated":"2025-05-20T13:03:15Z","snapshot_observed_at":"2026-08-08T01:19:21.267226Z","submitted_at":"2025-05-20T13:03:15Z","title":"Exploring Jailbreak Attacks on LLMs through Intent Concealment and Diversion","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-07T15:41:10.294985Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2505.14316"},"observation_digest":"sha256:870729633b5ddbb382614434166eb97db13d526d045a43563028ad87b24ca517","observation_id":"e7cfc789-826b-4fe4-8f1e-0e0814606f5e","resolution":{"observed_at":"2026-08-07T15:41:10.294985Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-07T15:09:57.250696Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17147","last_updated":"2025-05-22T08:22:57Z","snapshot_observed_at":"2026-08-08T03:05:46.469319Z","submitted_at":"2025-05-22T08:22:57Z","title":"MTSA: Multi-turn Safety Alignment for LLMs through Multi-round Red-teaming","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-07T15:09:57.250696Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2505.17147"},"observation_digest":"sha256:c96661447488c844feb28ba9c112386a5f7a6ebcb743eb2580c49d027fe11477","observation_id":"634fa79b-81ca-4dbe-9899-d1b732efd882","resolution":{"observed_at":"2026-08-07T15:09:57.250696Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-07T14:30:08.770366Z","title":"Red-teaming large language models using chain of utterances for safety-alignment","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.18807","last_updated":"2025-05-24T17:41:47Z","snapshot_observed_at":"2026-08-08T01:20:21.986624Z","submitted_at":"2025-05-24T17:41:47Z","title":"Mitigating Deceptive Alignment via Self-Monitoring","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:08.770366Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2505.18807"},"observation_digest":"sha256:dbe4f11cef58e1da8fdf31535adde2abcc1d9e89cc6b647bcafa4162c4679c15","observation_id":"46f8839c-2343-4faa-a81c-bff59ebf220b","resolution":{"observed_at":"2026-08-07T14:30:08.770366Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-07T14:27:03.407254Z","title":"Red-teaming large language mod- els using chain of utterances for safety-alignment","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.18889","last_updated":"2025-08-24T03:15:13Z","snapshot_observed_at":"2026-08-08T20:47:09.765405Z","submitted_at":"2025-05-24T22:22:43Z","title":"Security Concerns for Large Language Models: A Survey","version":5},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T14:27:03.407254Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2505.18889"},"observation_digest":"sha256:41c3c90b737c778f3e89f0f12ced99589b42ad0647519fa352091b179e843d16","observation_id":"cf7d8278-25e5-430d-8210-761a9ce1c0b0","resolution":{"observed_at":"2026-08-07T14:27:03.407254Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-07T14:11:30.658985Z","title":"Red-teaming large language models using chain of utterances for safety-alignment","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19743","last_updated":"2025-08-16T11:40:47Z","snapshot_observed_at":"2026-08-09T04:25:32.618058Z","submitted_at":"2025-05-26T09:24:36Z","title":"Token-level Accept or Reject: A Micro Alignment Approach for Large Language Models","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T14:11:30.658985Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2505.19743"},"observation_digest":"sha256:7be9454ea8e0bd56117a64fbf8b3cb4391f7c389de2ccd41888c23c29f905301","observation_id":"1d45dbd4-4bca-4e1b-ba73-570d344eedba","resolution":{"observed_at":"2026-08-07T14:11:30.658985Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-07T14:15:22.576261Z","title":"Red-teaming large language models using chain of utterances for safety-alignment","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.20359","last_updated":"2025-05-29T13:19:08Z","snapshot_observed_at":"2026-08-09T00:58:00.522901Z","submitted_at":"2025-05-26T08:01:37Z","title":"Risk-aware Direct Preference Optimization under Nested Risk Measure","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:22.576261Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2505.20359"},"observation_digest":"sha256:62e242178261182dbd61f3ca3d1175d8d2b16a7b737e2afdf520f611f2cfcd70","observation_id":"be60ffa7-ae5f-4129-a599-cc034230db8d","resolution":{"observed_at":"2026-08-07T14:15:22.576261Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-07T12:42:11.476447Z","title":"Red-teaming large language models using chain of utterances for safety-alignment","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.24019","last_updated":"2025-05-29T21:39:08Z","snapshot_observed_at":"2026-08-07T20:58:14.169872Z","submitted_at":"2025-05-29T21:39:08Z","title":"LLM Agents Should Employ Security Principles","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T12:42:11.476447Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2505.24019"},"observation_digest":"sha256:66e4bbf5e0e77dea6ddfc606c53c64554768677ea5ed9af1d3b36f7f2d498747","observation_id":"dbcb2b21-ad46-4d62-a163-76dcfe37b233","resolution":{"observed_at":"2026-08-07T12:42:11.476447Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-07T12:35:20.556699Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.24369","last_updated":"2025-05-30T09:02:07Z","snapshot_observed_at":"2026-08-09T16:12:29.442194Z","submitted_at":"2025-05-30T09:02:07Z","title":"Adversarial Preference Learning for Robust LLM Alignment","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-07T12:35:20.556699Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2505.24369"},"observation_digest":"sha256:849fc85546acba49e56ed10412383375e5c5093dc946951a0948acdf930be0a5","observation_id":"78c1836b-0f38-4111-aa6c-981b6d6ae984","resolution":{"observed_at":"2026-08-07T12:35:20.556699Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-07T05:35:09.998719Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10022","last_updated":"2025-06-09T12:02:39Z","snapshot_observed_at":"2026-08-07T05:25:53.532490Z","submitted_at":"2025-06-09T12:02:39Z","title":"LLMs Caught in the Crossfire: Malware Requests and Jailbreak Challenges","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T05:35:09.998719Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2506.10022"},"observation_digest":"sha256:4c1c17c6ec6a9d9060f8fa462834cd77f668fabba2814e82a84adf9bcd1fd145","observation_id":"ece555ab-f3a1-4a6a-89d4-87c87c5497b5","resolution":{"observed_at":"2026-08-07T05:35:09.998719Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-07T10:17:26.657917Z","title":"Red-teaming large language mod- els using chain of utterances for safety-alignment,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.11094","last_updated":"2025-10-30T06:22:33Z","snapshot_observed_at":"2026-08-07T10:11:06.747781Z","submitted_at":"2025-06-06T05:50:50Z","title":"The Scales of Justitia: A Comprehensive Survey on Safety Evaluation of LLMs","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T10:17:26.657917Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2506.11094"},"observation_digest":"sha256:89005ac513d783a8ca70abc6047cb465c24c2e3a7272f6ed450fd83b88771206","observation_id":"73959c27-4331-4ad2-ac15-f280e44a1f53","resolution":{"observed_at":"2026-08-07T10:17:26.657917Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-07T01:06:08.903210Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.12148","last_updated":"2025-06-13T18:08:19Z","snapshot_observed_at":"2026-08-09T12:18:43.237679Z","submitted_at":"2025-06-13T18:08:19Z","title":"Hatevolution: What Static Benchmarks Don't Tell Us","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-07T01:06:08.903210Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2506.12148"},"observation_digest":"sha256:2f6e271b1633e77ff62e27fdd96c69d16c291e2a67fdd59fc14c9d32269b6f8b","observation_id":"c2c388da-08b8-437f-9576-e73b9fee28e6","resolution":{"observed_at":"2026-08-07T01:06:08.903210Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-06T23:49:52.753276Z","title":"arXiv preprint arXiv:2308.09662","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.16322","last_updated":"2025-06-19T13:56:41Z","snapshot_observed_at":"2026-08-07T20:46:18.050527Z","submitted_at":"2025-06-19T13:56:41Z","title":"PL-Guard: Benchmarking Language Model Safety for Polish","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-06T23:49:52.753276Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2506.16322"},"observation_digest":"sha256:6af3ba2901e249b3d49f8b4e34fda54a3c18f46893fab82d0e006d9ab5c072b4","observation_id":"2e1182d2-c42d-48c9-8d81-b7028e4a16c1","resolution":{"observed_at":"2026-08-06T23:49:52.753276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-06T20:45:53.245262Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02057","last_updated":"2025-07-02T18:00:49Z","snapshot_observed_at":"2026-08-09T00:37:18.838001Z","submitted_at":"2025-07-02T18:00:49Z","title":"MGC: A Compiler Framework Exploiting Compositional Blindness in Aligned LLMs for Malware Generation","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T20:45:53.245262Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2507.02057"},"observation_digest":"sha256:0bba4e5e3f54c37e17281d0fc36b5bbe8dfeb59556028a96ee6bf96fc04f2bdb","observation_id":"3fe0b9fa-78e1-43e1-ac5d-954235d63b18","resolution":{"observed_at":"2026-08-06T20:45:53.245262Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-06T19:18:42.524938Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:42.524938Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:55333e6c67b337bb82d38c4bfffaae197c17260240b6ba5058f23063a23e4633","observation_id":"dca1a89b-0b75-4bdf-9f69-555eb9834939","resolution":{"observed_at":"2026-08-06T19:18:42.524938Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-06T17:17:27.596750Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11316","last_updated":"2025-07-15T13:48:35Z","snapshot_observed_at":"2026-08-08T00:39:39.784862Z","submitted_at":"2025-07-15T13:48:35Z","title":"Internal Value Alignment in Large Language Models through Controlled Value Vector Activation","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-06T17:17:27.596750Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2507.11316"},"observation_digest":"sha256:6a3bd15d1c8237dc7530430525c4e175054e35d3b74a837cd8143ce69cd1bc78","observation_id":"48420487-1b02-4ebc-8055-a103043207ed","resolution":{"observed_at":"2026-08-06T17:17:27.596750Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-06T17:16:33.404516Z","title":"Red-teaming large language models using chain of utterances for safety-alignment, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11344","last_updated":"2025-07-15T14:20:23Z","snapshot_observed_at":"2026-08-08T00:39:15.872422Z","submitted_at":"2025-07-15T14:20:23Z","title":"Guiding LLM Decision-Making with Fairness Reward Models","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-06T17:16:33.404516Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2507.11344"},"observation_digest":"sha256:d7369c1b0c20652abf4e82a8582238af93361327e3671be6dcb7df217f60b57a","observation_id":"dc893af8-465b-4303-8687-e6e0658b8567","resolution":{"observed_at":"2026-08-06T17:16:33.404516Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T17:26:48.677262Z","title":"Red-teaming large language models using chain of utterances for safety-alignment.arXiv preprint arXiv:2308.09662, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.16318","last_updated":"2025-09-01T08:35:27Z","snapshot_observed_at":"2026-08-07T01:49:06.395378Z","submitted_at":"2025-08-22T11:57:55Z","title":"SATORI: Static Test Oracle Generation for REST APIs","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-05T17:26:48.677262Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2508.16318"},"observation_digest":"sha256:7dbe4e9ffe680cba45e425c10bd54b53ba4024c24fac442267f7beedf76d5609","observation_id":"e31d8f53-d82f-4ae8-8d80-108c0eabed2d","resolution":{"observed_at":"2026-08-05T17:26:48.677262Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2509.10546","last_updated":"2026-04-24T18:29:01Z","snapshot_observed_at":"2026-08-02T04:58:06.974646Z","submitted_at":"2025-09-07T22:35:15Z","title":"Learning to Conceal Risk: Controllable Multi-turn Red Teaming for LLMs in the Financial Domain","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-05-18T17:49:42.112564Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2509.10546"},"observation_digest":"sha256:b97bd9e01922cada8dce73fca0255be4c06924c943c64072853ec559e7aecdce","observation_id":"4b6e49a8-3215-48d8-8cea-84d2b7846095","resolution":{"observed_at":"2026-05-18T17:51:41.967482Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2509.11206","last_updated":"2026-04-20T05:43:30Z","snapshot_observed_at":"2026-08-07T05:35:56.514953Z","submitted_at":"2025-09-14T10:24:13Z","title":"Evalet: Evaluating Large Language Models through Functional Fragmentation","version":4},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-18T16:57:25.259866Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2509.11206"},"observation_digest":"sha256:2ff346e3802b69cb0b0f14e486baab9eeed3f8ebc41ab8b852f85be625535067","observation_id":"bae77025-96af-4599-a740-cfa343fa042e","resolution":{"observed_at":"2026-05-18T17:01:40.065888Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2510.09689","last_updated":"2026-04-17T02:42:09Z","snapshot_observed_at":"2026-07-06T22:32:22.953630Z","submitted_at":"2025-10-09T09:44:14Z","title":"When Search Goes Wrong: Red-Teaming Web-Augmented Large Language Models","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-18T09:29:14.842228Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2510.09689"},"observation_digest":"sha256:dd6f098ec6f322cd2aa33c639a77b7d60edb3f1d8825847fbfd1ba7a8a1ae13b","observation_id":"d4b5a365-d421-46b2-b351-7573e17cf87c","resolution":{"observed_at":"2026-05-18T09:31:11.600666Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-04T09:25:36.898264Z","title":"Red-teaming large language models using chain of utterances for safety-alignment","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2510.15476","last_updated":"2026-07-04T04:20:15Z","snapshot_observed_at":"2026-08-06T02:45:04.316908Z","submitted_at":"2025-10-17T09:38:54Z","title":"SoK: Systematizing LLM Prompt Security: Taxonomies, Datasets, and Unified Evaluation of Attacks and Defenses","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-04T09:25:36.898264Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2510.15476"},"observation_digest":"sha256:e86f13617f7dc744e5c74a144b8bbf7be9d9e4c8ff7650ad06bcec5f58db532b","observation_id":"e41f6a8a-e9f9-4042-9641-5bcc6b2df7a1","resolution":{"observed_at":"2026-08-04T09:25:36.898264Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-03T22:03:17.979326Z","title":"ArXivabs/2308.09662(2023),https://api","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2511.12487","last_updated":"2026-01-24T06:07:58Z","snapshot_observed_at":"2026-08-06T03:06:07.028822Z","submitted_at":"2025-11-16T07:47:31Z","title":"ToxSearch: Evolving Prompts for Toxicity Search in Large Language Models","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-03T22:03:17.979326Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2511.12487"},"observation_digest":"sha256:dccd720295ce3aa18e01853355777199e2b737dda4e0407fa07960a62eaf692b","observation_id":"3c6a59d3-cb9a-4631-be2b-e88e2783613e","resolution":{"observed_at":"2026-08-03T22:03:17.979326Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2601.17887","last_updated":"2026-05-17T15:59:35Z","snapshot_observed_at":"2026-08-02T19:18:47.080046Z","submitted_at":"2026-01-25T15:42:01Z","title":"When Personalization Legitimizes Risks: Uncovering Safety Vulnerabilities in Personalized Dialogue Agents","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-21T15:11:04.394636Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2601.17887"},"observation_digest":"sha256:e521aafc867adb77ea19ac3e34ab976bab679e24b03d0ff5959c41db3f8432db","observation_id":"7e1d4702-9bb7-43e7-bb08-5e519c59d389","resolution":{"observed_at":"2026-05-21T15:14:13.377473Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2601.20981","last_updated":"2026-04-21T09:20:29Z","snapshot_observed_at":"2026-08-03T04:22:47.507604Z","submitted_at":"2026-01-28T19:29:54Z","title":"Diversifying Toxicity Search in Large Language Models Through Speciation","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-16T09:40:40.955703Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2601.20981"},"observation_digest":"sha256:ca9a9094fa80ff613515db4d0693dd85277bfd86482a84ffadb7a72143124279","observation_id":"ea431d66-5646-44b9-8f16-ae81d0743128","resolution":{"observed_at":"2026-05-16T09:40:48.715471Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2605.01687","last_updated":"2026-05-03T02:55:30Z","snapshot_observed_at":"2026-07-06T23:14:52.417213Z","submitted_at":"2026-05-03T02:55:30Z","title":"MultiBreak: A Scalable and Diverse Multi-turn Jailbreak Benchmark for Evaluating LLM Safety","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-05-10T16:00:32.413225Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2605.01687"},"observation_digest":"sha256:4c1328e718158ca035ddefbea3587973d642c5a816ad5e4fb812a22581875cf2","observation_id":"8e386a28-aaa2-4d51-9187-1f729c03705b","resolution":{"observed_at":"2026-05-11T09:31:01.284910Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2605.04446","last_updated":"2026-05-06T03:21:38Z","snapshot_observed_at":"2026-07-06T23:17:13.770090Z","submitted_at":"2026-05-06T03:21:38Z","title":"Misrouter: Exploiting Routing Mechanisms for Input-Only Attacks on Mixture-of-Experts LLMs","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-08T18:07:00.247161Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2605.04446"},"observation_digest":"sha256:ea28fbca567ea159904919dae7ab5353b2356a9babbee37065d0e88e9ecb1c17","observation_id":"65f6e4cf-e496-4a14-98f8-2f9057019a7c","resolution":{"observed_at":"2026-05-09T06:45:44.444934Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2605.04992","last_updated":"2026-05-06T14:52:22Z","snapshot_observed_at":"2026-07-06T23:17:43.402046Z","submitted_at":"2026-05-06T14:52:22Z","title":"You Snooze, You Lose: Automatic Safety Alignment Restoration through Neural Weight Translation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-08T17:02:20.836208Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2605.04992"},"observation_digest":"sha256:58b4b374f92151ae6f5b08f65f48e924c4af68279e5c1198713c35a363e77cad","observation_id":"c7778730-89ca-470d-b42a-0ffd009ed8b2","resolution":{"observed_at":"2026-05-11T17:51:08.429806Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2605.19190","last_updated":"2026-05-18T23:34:53Z","snapshot_observed_at":"2026-08-02T02:36:37.244109Z","submitted_at":"2026-05-18T23:34:53Z","title":"Going PLACES: Participatory Localized Red Teaming for Text-to-Image Safety in the Global South","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-20T07:06:57.555070Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2605.19190"},"observation_digest":"sha256:f3454e7adb770c607577ebeb15cd5302204760b6bccb9f598fe05ce3bef29f8c","observation_id":"6d3603b4-cad3-4497-ae07-151bd157bf72","resolution":{"observed_at":"2026-05-20T07:08:07.009492Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2606.00686","last_updated":"2026-05-30T11:49:42Z","snapshot_observed_at":"2026-08-04T22:47:09.122420Z","submitted_at":"2026-05-30T11:49:42Z","title":"Dialectics of Alignment: Harnessing Unsafe Knowledge for Dynamic Safety Routing","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-28T19:23:10.275977Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2606.00686"},"observation_digest":"sha256:8bb0d0538b6d034636ffff70e797dd4dbb125bf3132b271848ebce32ceec40ce","observation_id":"6991985d-2de9-49f6-bd3c-a95fd28a1d48","resolution":{"observed_at":"2026-06-28T19:32:34.566985Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2606.02530","last_updated":"2026-06-01T17:38:12Z","snapshot_observed_at":"2026-08-05T11:28:53.105574Z","submitted_at":"2026-06-01T17:38:12Z","title":"SafeSteer: Localized On-Policy Distillation for Efficient Safety Alignment","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-06-28T14:39:11.178976Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2606.02530"},"observation_digest":"sha256:b65c715291aa07917a7b97a09f017fd4735f600f8d93d355fbaa7361a65c5280","observation_id":"25061803-f922-4013-9920-04f853853a18","resolution":{"observed_at":"2026-07-01T23:06:20.835743Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2606.07335","last_updated":"2026-06-05T14:49:26Z","snapshot_observed_at":"2026-07-06T23:47:00.425715Z","submitted_at":"2026-06-05T14:49:26Z","title":"Defending Jailbreak Attacks on Large Language Models via Manifold Trajectory Kinetics","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-06-27T21:55:48.561400Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2606.07335"},"observation_digest":"sha256:9c22582e8a2e01d4054f7773256b991f59a550d89b5b5c43e641c9d804b4ac3c","observation_id":"2dba2dfc-6332-40b8-9f6a-2f3f983906c9","resolution":{"observed_at":"2026-07-02T17:37:14.894469Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2606.24166","last_updated":"2026-06-23T05:40:05Z","snapshot_observed_at":"2026-07-06T23:58:49.574536Z","submitted_at":"2026-06-23T05:40:05Z","title":"Distributed Quality-Diversity Search for Toxicity in Large Language Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-25T22:12:27.301100Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2606.24166"},"observation_digest":"sha256:75f5094387c5dab572541253518a03d5dd73909e91e82ad9d3efb30f8eede74a","observation_id":"587eacad-b34c-4eed-ae8e-424a3e580995","resolution":{"observed_at":"2026-07-04T19:00:04.582802Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-04T01:16:12.719886Z","title":"Red-teaming large language mod- els using chain of utterances for safety-alignment,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.00134","last_updated":"2026-07-31T14:04:49Z","snapshot_observed_at":"2026-08-07T20:59:09.642310Z","submitted_at":"2026-07-31T14:04:49Z","title":"Stateful Cooperative Agents Safeguarding LLMs Against Evolving Multi-Turn Attacks","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-04T01:16:12.719886Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2608.00134"},"observation_digest":"sha256:0f3cfd337fb91123ef33d587370e7627a51739f15d82aad1b98941c193fee896","observation_id":"5aa005cc-1aac-45f5-b27f-407bd964ef24","resolution":{"observed_at":"2026-08-04T01:16:12.719886Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-06T00:18:48.516596Z","title":"Red-teaming large language models using chain of utterances for safety-alignment.arXiv preprint arXiv:2308.09662, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.01414","last_updated":"2026-08-02T17:49:07Z","snapshot_observed_at":"2026-08-08T23:34:22.683191Z","submitted_at":"2026-08-02T17:49:07Z","title":"No Single Neuron of Failure: Distributed Safety Alignment Against White-Box Attacks","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T00:18:48.516596Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2608.01414"},"observation_digest":"sha256:a551b607dc96e6871b1b1c8c2420120264d8c0fbf3432d4fcf1e603172538355","observation_id":"a8e4ed37-5070-4455-b867-09c47f93ea5b","resolution":{"observed_at":"2026-08-06T00:18:48.516596Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-06T00:29:56.632621Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.03166","last_updated":"2026-08-04T05:54:32Z","snapshot_observed_at":"2026-08-08T00:04:44.793928Z","submitted_at":"2026-08-04T05:54:32Z","title":"Adversarial Stress Testing of Role-Playing Language Agents using Multi-Agent Evaluation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T00:29:56.632621Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2608.03166"},"observation_digest":"sha256:bf6b8c555feb580f494efb48991af27accc93585e45b3d4d102d95567a778b09","observation_id":"51ee87da-f0b5-42b3-9169-f8cfd2021453","resolution":{"observed_at":"2026-08-06T00:29:56.632621Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2308.09662/citation-record","integrity":"/paper/2308.09662/integrity","json":"/paper/2308.09662/citation-record.json","paper":"/paper/2308.09662"},"outbound":[],"paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","latest_version":3,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 39 inbound Pith citation observations for arXiv:2308.09662."}