{"as_of":"2026-08-14T01:31:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:23f1965c8ace3359cc65a8ff4cf79798a45b91edfcbd150dae8f2875f10e82c9","coverage":[{"denominator":58,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":58,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:15:48.828674Z","state":"measured"},{"denominator":61,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":61,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T18:55:01.175611Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-19T11:37:15.922072Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"cited_work":{"arxiv_id":"2505.15753","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15753","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Scalable defense against in-the-wild jailbreaking attacks with safety context retrieval","venue":null,"work_id":"961e2bad-cd86-4f80-a9e2-1883400215e3","year":2025},"citing_paper":{"arxiv_id":"2506.01770","last_updated":"2026-04-18T05:40:12Z","snapshot_observed_at":"2026-08-04T01:20:56.393646Z","submitted_at":"2025-06-02T15:17:38Z","title":"ReGA: Model-Based Safeguard for LLMs via Representation-Guided Abstraction","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-19T11:34:09.428653Z"},"links":{"cited_paper":"/paper/2505.15753","citing_paper":"/paper/2506.01770"},"observation_digest":"sha256:6baf50c6cdde4ce1c5ae1fe71f665caa69ab6d3154a72e393857f968fff87ad3","observation_id":"d0e4b5aa-4b40-4323-bfb8-6848d43a9507","resolution":{"observed_at":"2026-05-19T11:37:15.923828Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"cited_work":{"arxiv_id":"2505.15753","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15753","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Scalable defense against in-the-wild jailbreaking attacks with safety context retrieval","venue":null,"work_id":"961e2bad-cd86-4f80-a9e2-1883400215e3","year":2025},"citing_paper":{"arxiv_id":"2512.12069","last_updated":"2026-04-20T07:36:16Z","snapshot_observed_at":"2026-08-11T09:56:37.609830Z","submitted_at":"2025-12-12T22:31:38Z","title":"Rethinking Jailbreak Detection of Large Vision Language Models with Representational Contrastive Scoring","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-16T22:28:41.134253Z"},"links":{"cited_paper":"/paper/2505.15753","citing_paper":"/paper/2512.12069"},"observation_digest":"sha256:c5289b45e3d4de0d10ee82ecf5cf6e08afdc821ee2075b06357d77deb703acb5","observation_id":"8d7f6248-08d0-45b0-8e87-2bc492581934","resolution":{"observed_at":"2026-05-16T22:31:19.445142Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15753","snapshot_observed_at":"2026-08-01T18:55:01.175611Z","title":"arXiv preprint arXiv:2505.15753 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.17152","last_updated":"2026-07-19T09:18:35Z","snapshot_observed_at":"2026-08-09T14:18:14.468847Z","submitted_at":"2026-07-19T09:18:35Z","title":"How Jailbreak Attacks Inform Safety Alignment: A Defender-Centric, Shapley-Based Evaluation of Jailbreak Contributions","version":1},"reference_index":101,"source":"arxiv_source","source_observed_at":"2026-08-01T18:55:01.175611Z"},"links":{"cited_paper":"/paper/2505.15753","citing_paper":"/paper/2607.17152"},"observation_digest":"sha256:6487f7df9b74193da7bdd247f51a9b139da6e05f64951449b320736fa77de46f","observation_id":"a6d4da1b-bd9f-48fb-aa91-94f1b4938306","resolution":{"observed_at":"2026-08-01T18:55:01.175611Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2505.15753/citation-record","integrity":"/paper/2505.15753/integrity","json":"/paper/2505.15753/citation-record.json","paper":"/paper/2505.15753"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2308.14132","last_updated":"2023-11-07T03:30:15Z","snapshot_observed_at":"2026-07-06T16:10:54.723336Z","submitted_at":"2023-08-27T15:20:06Z","title":"Detecting Language Model Attacks with Perplexity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.14132","snapshot_observed_at":"2026-08-07T15:15:48.522821Z","title":"Detecting language model attacks with perplexity","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.522821Z"},"links":{"cited_paper":"/paper/2308.14132","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:3731110ca279344f23df683ccc46cc8a2e8c99009ca236722dec1a29d59a846c","observation_id":"99b7ab13-ce2b-460a-adf7-242f3bd5d271","resolution":{"observed_at":"2026-08-07T15:15:48.522821Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.18041","last_updated":"2025-04-25T03:25:18Z","snapshot_observed_at":"2026-08-12T21:18:58.152317Z","submitted_at":"2025-04-25T03:25:18Z","title":"RAG LLMs are Not Safer: A Safety Analysis of Retrieval-Augmented Generation for Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.18041","snapshot_observed_at":"2026-08-07T15:15:48.529201Z","title":"Rag llms are not safer: A safety analysis of retrieval-augmented generation for large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.529201Z"},"links":{"cited_paper":"/paper/2504.18041","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:ea600e4889e5259eb35111668649ed0711809e27aaa3e3db1366179fc1ee1085","observation_id":"bbb68cd4-3423-4fb7-946d-878e2646fb5f","resolution":{"observed_at":"2026-08-07T15:15:48.529201Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:52.488427Z","title":"Qwen technical report","venue":null,"work_id":"01b884a8-7eca-49fa-ae00-58c6f721c16b","year":2023},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.535290Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:8a2e89ca7c9aafa8ebb4b03bcd8da9567bfbaea7e3644c68760581713c42267b","observation_id":"c64ceb1b-9e7c-49c2-a707-4a12513aec4c","resolution":{"observed_at":"2026-08-07T15:15:52.609923Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:48.540149Z","title":"Constitutional ai: Harmlessness from ai feedback, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.540149Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:58b79148c17d8f0e852b5125f8accea8c5e11a2fccd810f4741930ef6b2aa409","observation_id":"886a8169-04ba-48e4-b2e7-e55a244fca48","resolution":{"observed_at":"2026-08-07T15:15:48.540149Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:52.323434Z","title":"Safeinfer: Context adaptive decoding time safety alignment for large language models","venue":null,"work_id":"cc53d270-2812-4e4b-a67a-cc9d55f0c9cd","year":2025},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.544775Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:b3872b3cf48104de62cad4310cb30878a71cae27a89841a9046caf6524def5e1","observation_id":"b45cab46-1d09-4a98-879c-4da8a4448c56","resolution":{"observed_at":"2026-08-07T15:15:52.384358Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08419","last_updated":"2024-07-18T18:24:57Z","snapshot_observed_at":"2026-08-12T12:29:45.113982Z","submitted_at":"2023-10-12T15:38:28Z","title":"Jailbreaking Black Box Large Language Models in Twenty Queries","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.08419","snapshot_observed_at":"2026-08-07T15:15:48.550190Z","title":"Jailbreaking black box large language models in twenty queries","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.550190Z"},"links":{"cited_paper":"/paper/2310.08419","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:2eca957bf3aaebc6d97736d56ce6bb6dab7d445bc42b65490415c5465423fe6e","observation_id":"98ec7c4c-dc42-4b15-8d62-997bdb9eb3a4","resolution":{"observed_at":"2026-08-07T15:15:48.550190Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:48.555459Z","title":"Towards the worst-case robustness of large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.555459Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:09a31eb936e7eb6877fb9bfa273b62451738f6a9f04bb3177202e59ec00cb613","observation_id":"5d460c8a-66b5-425c-ac4e-76febce85e8c","resolution":{"observed_at":"2026-08-07T15:15:48.555459Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:51.970388Z","title":"Evaluating large language models trained on code, 2021","venue":null,"work_id":"a356a68e-d836-49f7-b59d-1144751be8cc","year":2021},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.559854Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:5ccda344ab94a164e111b345980af2e96a0916bc0fbb0c231956092f4021a1be","observation_id":"d1712231-c99f-48e3-a1a8-e34fb7c9b811","resolution":{"observed_at":"2026-08-07T15:15:52.170794Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-07T15:15:48.565245Z","title":"Training verifiers to solve math word problems","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.565245Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:5c23336c2aa886bb6cc7f438991cd80bb1f00e678898e9c4dbef297e60cc3ffb","observation_id":"3db68c66-159c-4920-bc1a-6c8fe9a743c8","resolution":{"observed_at":"2026-08-07T15:15:48.565245Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:51.540719Z","title":"Safe rlhf: Safe reinforcement learning from human feedback","venue":null,"work_id":"c82ca8f4-9065-447b-96f2-1ffdcfb417c0","year":2024},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.571116Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:4863a6bfb4d20a14c7fdc12be82e9ec5cc2ab6f09f689c16b1d3ba2da47fd6aa","observation_id":"00503b70-36cd-4723-bc38-02c821a144c7","resolution":{"observed_at":"2026-08-07T15:15:51.667541Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06474","last_updated":"2024-03-04T04:03:54Z","snapshot_observed_at":"2026-08-13T05:53:10.332594Z","submitted_at":"2023-10-10T09:44:06Z","title":"Multilingual Jailbreak Challenges in Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06474","snapshot_observed_at":"2026-08-07T15:15:48.577618Z","title":"Multilingual jailbreak chal- lenges in large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.577618Z"},"links":{"cited_paper":"/paper/2310.06474","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:0da98b855abb260dd35aa9882b5d8cedacb7388538b532150970b253b7d42a0b","observation_id":"316fc115-9d99-4232-b4e7-4d35eac70880","resolution":{"observed_at":"2026-08-07T15:15:48.577618Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.08268","last_updated":"2024-04-07T03:04:10Z","snapshot_observed_at":"2026-08-13T05:25:30.661845Z","submitted_at":"2023-11-14T16:02:16Z","title":"A Wolf in Sheep's Clothing: Generalized Nested Jailbreak Prompts can Fool Large Language Models Easily","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.08268","snapshot_observed_at":"2026-08-07T15:15:48.584004Z","title":"A wolf in sheep’s clothing: Generalized nested jailbreak prompts can fool large language models easily","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.584004Z"},"links":{"cited_paper":"/paper/2311.08268","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:d00a4910a634a96175b71efd45abc32d23b480ae9517733e4def4a5f42c93007","observation_id":"a248c4bd-a27b-4259-ab97-eee247e8dff8","resolution":{"observed_at":"2026-08-07T15:15:48.584004Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.10997","last_updated":"2024-03-27T09:16:57Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-18T07:47:33Z","title":"Retrieval-Augmented Generation for Large Language Models: A Survey","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.10997","snapshot_observed_at":"2026-08-07T15:15:48.589805Z","title":"Retrieval-augmented generation for large language models: A survey","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.589805Z"},"links":{"cited_paper":"/paper/2312.10997","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:bd8957482597e4daf0b8fc15e4dc427c954341661a4744da191a289eaabc9e25","observation_id":"1e1a44d9-31bf-422e-b958-1cae8ab91900","resolution":{"observed_at":"2026-08-07T15:15:48.589805Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1412.6572","last_updated":"2015-03-20T20:19:16Z","snapshot_observed_at":"2026-08-12T17:13:46.394331Z","submitted_at":"2014-12-20T01:17:12Z","title":"Explaining and Harnessing Adversarial Examples","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1412.6572","snapshot_observed_at":"2026-08-07T15:15:48.594740Z","title":"Goodfellow, Jonathon Shlens, and Christian Szegedy","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.594740Z"},"links":{"cited_paper":"/paper/1412.6572","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:ccc2ba8539bbfbd0acde5faef67c413c82379bf9db2939d049b0fd69725379e0","observation_id":"f58024da-2a64-4c2a-adc5-c54e21b63d16","resolution":{"observed_at":"2026-08-07T15:15:48.594740Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-07T15:15:48.601205Z","title":"The llama 3 herd of models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.601205Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:d81d0adc64b4244713ea482cd6b124a55e1fc8c0f2e3c68178413634f812efcc","observation_id":"5c6df03e-c5fc-4ac7-808a-95433d683301","resolution":{"observed_at":"2026-08-07T15:15:48.601205Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:48.605873Z","title":"Measuring massive multitask language understanding","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.605873Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:209cc8774e932ed4d58aed39ab5353bf9c1d55444355d79f84abba0ec745a34f","observation_id":"5520535c-e1fb-4184-9649-3fd34870b845","resolution":{"observed_at":"2026-08-07T15:15:48.605873Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.06674","last_updated":"2023-12-07T19:40:50Z","snapshot_observed_at":"2026-08-13T00:51:41.824778Z","submitted_at":"2023-12-07T19:40:50Z","title":"Llama Guard: LLM-based Input-Output Safeguard for Human-AI Conversations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.06674","snapshot_observed_at":"2026-08-07T15:15:48.609882Z","title":"Llama guard: Llm-based input-output safeguard for human-ai conversations","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.609882Z"},"links":{"cited_paper":"/paper/2312.06674","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:4cd138d52b2110cba7744906ec925c52fabfdb97c8c52ab1e439535cf360cec9","observation_id":"52d6f016-d02b-4072-953a-d445c93b68cd","resolution":{"observed_at":"2026-08-07T15:15:48.609882Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.00614","last_updated":"2023-09-04T17:47:36Z","snapshot_observed_at":"2026-07-06T16:13:23.343694Z","submitted_at":"2023-09-01T17:59:44Z","title":"Baseline Defenses for Adversarial Attacks Against Aligned Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.00614","snapshot_observed_at":"2026-08-07T15:15:48.615884Z","title":"Baseline de- fenses for adversarial attacks against aligned language models.arXiv preprint arXiv:2309.00614, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.615884Z"},"links":{"cited_paper":"/paper/2309.00614","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:ce979640a621706d67bb819c2a0358fbb50bfef608ba7f7f970d334b1cb410e9","observation_id":"c8206a2c-c27e-4148-b506-cd9e25c144b1","resolution":{"observed_at":"2026-08-07T15:15:48.615884Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.19852","last_updated":"2025-04-04T11:14:49Z","snapshot_observed_at":"2026-08-05T20:02:35.087707Z","submitted_at":"2023-10-30T15:52:15Z","title":"AI Alignment: A Comprehensive Survey","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.19852","snapshot_observed_at":"2026-08-07T15:15:48.621390Z","title":"Ai alignment: A comprehensive survey","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.621390Z"},"links":{"cited_paper":"/paper/2310.19852","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:56feb3bb198a16c1efb545a6b5c6b2c81400d5f0f7bf9cd83db76b0384abf339","observation_id":"f64e1a99-04dc-4715-9a5d-43ae87f3f4c1","resolution":{"observed_at":"2026-08-07T15:15:48.621390Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.21018","last_updated":"2024-06-05T16:35:49Z","snapshot_observed_at":"2026-08-12T23:53:12.380438Z","submitted_at":"2024-05-31T17:07:15Z","title":"Improved Techniques for Optimization-Based Jailbreaking on Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.21018","snapshot_observed_at":"2026-08-07T15:15:48.626254Z","title":"Improved techniques for optimization-based jailbreaking on large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.626254Z"},"links":{"cited_paper":"/paper/2405.21018","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:05556410c4774089fdedbd79b1b23e6c203a2d2835395c0ee6c1224c0250cf3e","observation_id":"8729fdfa-68d5-4471-96b8-893b157485b5","resolution":{"observed_at":"2026-08-07T15:15:48.626254Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06825","last_updated":"2023-10-10T17:54:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T17:54:58Z","title":"Mistral 7B","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06825","snapshot_observed_at":"2026-08-07T15:15:48.631589Z","title":"Mistral 7b","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.631589Z"},"links":{"cited_paper":"/paper/2310.06825","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:65311362c382e33e2b37edd61e8d45c23cd5a1e8750c4a63e75fbc1908633e21","observation_id":"1995933e-f583-4470-a0a7-35dbd5bd062b","resolution":{"observed_at":"2026-08-07T15:15:48.631589Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:48.636714Z","title":"Artprompt: Ascii art-based jailbreak attacks against aligned llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.636714Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:3b193e663af05840b1c2176cb64649c3ab58f9ca40d25d20f3dbf515dee8539e","observation_id":"31bca02c-52af-4830-80f1-59a011fa3116","resolution":{"observed_at":"2026-08-07T15:15:48.636714Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:48.641406Z","title":"Wildteaming at scale: From in-the-wild jailbreaks to (adversarially) safer language models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.641406Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:0d7fe30f03d21bb2db803875ed677a3a4e7fb2b599179f79019baead2a8a749d","observation_id":"14b50c3f-e2d8-42a3-b948-a0a6529ea65f","resolution":{"observed_at":"2026-08-07T15:15:48.641406Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:48.646827Z","title":"Dense passage retrieval for open-domain question answering","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.646827Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:37553af4d629328b47860745b7d925b6f28ab5beb74c17354fc67ee82c4091cf","observation_id":"b40304df-a0f8-432a-a08a-754a4a82db50","resolution":{"observed_at":"2026-08-07T15:15:48.646827Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:51.215732Z","title":"Buckley, Jason Phang, Samuel R","venue":null,"work_id":"1a6a8070-2896-47b4-8bc6-b3e52cf03df1","year":2023},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.651724Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:8fea3f42bbde1c657f93c5f7a21382b2e428d5560c57c90557cf7941818af870","observation_id":"1c615d12-22ed-4350-9a2f-985701feaae0","resolution":{"observed_at":"2026-08-07T15:15:51.316907Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.01446","last_updated":"2024-08-05T11:34:10Z","snapshot_observed_at":"2026-08-13T10:20:58.033243Z","submitted_at":"2023-09-04T08:54:20Z","title":"Open Sesame! Universal Black Box Jailbreaking of Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.01446","snapshot_observed_at":"2026-08-07T15:15:48.657664Z","title":"Open sesame! universal black box jailbreaking of large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.657664Z"},"links":{"cited_paper":"/paper/2309.01446","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:58715cc8068f5c92bd25462e684f4b0d30a350a1c55b8b0ba71b8dcf1863ed82","observation_id":"5df0e263-d0cd-4aa3-ae20-573e45d7e151","resolution":{"observed_at":"2026-08-07T15:15:48.657664Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:48.663277Z","title":"Retrieval-augmented generation for knowledge-intensive nlp tasks","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.663277Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:46fe0120ce668ca0e330b28933926b18ce9f1adb05f0209260e7d23a688f6e29","observation_id":"f0e0be59-58ec-4d49-8f45-4ca9174e2b49","resolution":{"observed_at":"2026-08-07T15:15:48.663277Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.03191","last_updated":"2024-11-28T13:43:50Z","snapshot_observed_at":"2026-08-13T20:39:50.435115Z","submitted_at":"2023-11-06T15:29:30Z","title":"DeepInception: Hypnotize Large Language Model to Be Jailbreaker","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.03191","snapshot_observed_at":"2026-08-07T15:15:48.669536Z","title":"Deepinception: Hypnotize large language model to be jailbreaker","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.669536Z"},"links":{"cited_paper":"/paper/2311.03191","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:d9cc3ccda45881e5b01078ba948d3c07a4507d47c356839e3ad55011e167287e","observation_id":"61a32c2e-917c-475f-8bbb-020ed343ffbf","resolution":{"observed_at":"2026-08-07T15:15:48.669536Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.03281","last_updated":"2023-08-07T03:52:59Z","snapshot_observed_at":"2026-08-04T23:10:23.964516Z","submitted_at":"2023-08-07T03:52:59Z","title":"Towards General Text Embeddings with Multi-stage Contrastive Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.03281","snapshot_observed_at":"2026-08-07T15:15:48.674872Z","title":"Towards general text embeddings with multi-stage contrastive learning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.674872Z"},"links":{"cited_paper":"/paper/2308.03281","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:d516fb733bc85d02fa64ddd0002ec148eca60c20ba202de4b7e2596267aed96a","observation_id":"97c73d0b-3953-49fd-9b60-d83085a05ea3","resolution":{"observed_at":"2026-08-07T15:15:48.674872Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:48.679903Z","title":"Autodan: Generating stealthy jailbreak prompts on aligned large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.679903Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:8b64f8d8edc2aa2995f96a88d4572b3bbb79d4dc1a181d795c4469e3e7583c7f","observation_id":"21b586a7-03a1-453f-aca6-8674ddee56fb","resolution":{"observed_at":"2026-08-07T15:15:48.679903Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:51.070527Z","title":"Jailbreaking chatgpt via prompt engineering: An empirical study, 2023","venue":null,"work_id":"eb6f8ac9-ddeb-42d5-b7b3-85d60c93e4db","year":2023},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.685719Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:a92144243dec900e14edc98964d49434191db7c8c8160cb0e057f930bb6a9c81","observation_id":"417498db-a5c1-4c68-af90-e1fe302d606b","resolution":{"observed_at":"2026-08-07T15:15:51.112343Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:48.690350Z","title":"Harmbench: A standardized evaluation framework for automated red teaming and robust refusal","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.690350Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:da5463eecdd8b9acf1b70b9fb28c3a45c456433ef3afbad6b27ce792d7758a6b","observation_id":"3e2a2a90-9575-48b8-b1c4-1c77ed3fe5db","resolution":{"observed_at":"2026-08-07T15:15:48.690350Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:50.981966Z","title":"Tree of attacks: Jailbreaking black-box llms automatically.Advances in Neural Information Processing Systems , 37:61065–61105, 2024","venue":null,"work_id":"a54b0eb5-c97e-426e-b456-24828819e4b5","year":2024},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.695248Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:ede4d77334e902d8fa53a7fbe9371394cad2cff061ef250f7a0f73a2a56d61be","observation_id":"a42ad48c-47f4-4623-a68e-88ac9302a7cd","resolution":{"observed_at":"2026-08-07T15:15:51.015657Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.07494","last_updated":"2024-11-12T02:44:49Z","snapshot_observed_at":"2026-08-12T22:02:38.785237Z","submitted_at":"2024-11-12T02:44:49Z","title":"Rapid Response: Mitigating LLM Jailbreaks with a Few Examples","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.07494","snapshot_observed_at":"2026-08-07T15:15:48.700337Z","title":"Rapid response: Mitigating llm jailbreaks with a few examples","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.700337Z"},"links":{"cited_paper":"/paper/2411.07494","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:1ceeb4df98e609de35c32ab3f122461a478ee9219389abaf82b89f6115727fe5","observation_id":"f9dd6051-07a6-4c12-9962-15a75bfd8fe9","resolution":{"observed_at":"2026-08-07T15:15:48.700337Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.02260","last_updated":"2026-06-02T17:48:48Z","snapshot_observed_at":"2026-08-12T14:47:41.023865Z","submitted_at":"2025-02-04T12:17:08Z","title":"Position: Adversarial ML for LLMs Is Not Making Any Progress","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.02260","snapshot_observed_at":"2026-08-07T15:15:48.706076Z","title":"Adversarial ml problems are getting harder to solve and to evaluate","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.706076Z"},"links":{"cited_paper":"/paper/2502.02260","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:de92cdd177c5f1b8c221e2cd42b13f555a9f70571151bc95fd660659c66acbe9","observation_id":"38744573-a46d-4814-aaf5-834a00156e4a","resolution":{"observed_at":"2026-08-07T15:15:48.706076Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:50.871158Z","title":"The probabilistic relevance framework: Bm25 and beyond","venue":null,"work_id":"eb0e4a47-8732-4d96-8227-49845b4f4acb","year":2009},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.711127Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:ab2dadfce742155ef46e167d20b0b6fc758cdd4e69bdffb10608b3f8689aef46","observation_id":"13e1fa99-f5d6-49d1-923f-23fd1959b7b9","resolution":{"observed_at":"2026-08-07T15:15:50.918748Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:50.770355Z","title":"Mitigating skeleton key, a new type of generative ai jailbreak technique","venue":null,"work_id":"13aac4fd-6b8a-4bfc-b094-c4c8089f918d","year":2024},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.716170Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:c38b3cac6b4d36f0886ea2e5321d2c486d6b851d09b1beb25aff3a8c39deb648","observation_id":"2a0ec857-ce8b-41a3-875c-f07ea6f12cb2","resolution":{"observed_at":"2026-08-07T15:15:50.810370Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:48.720684Z","title":"Fine-tuning mistral 7b large language model for python query response and code generation: A parameter efficient approach","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.720684Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:0684410329d6a4407d7b76817c29d97aa86a0053893310c8c10937a6663de6f0","observation_id":"f15bcb96-d1a4-4fa6-8051-2dbd064f51c2","resolution":{"observed_at":"2026-08-07T15:15:48.720684Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1312.6199","last_updated":"2014-02-19T16:33:14Z","snapshot_observed_at":"2026-07-06T03:31:33.797310Z","submitted_at":"2013-12-21T03:36:08Z","title":"Intriguing properties of neural networks","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1312.6199","snapshot_observed_at":"2026-08-07T15:15:48.725718Z","title":"Intriguing properties of neural networks","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.725718Z"},"links":{"cited_paper":"/paper/1312.6199","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:7d121d34f5ff443624e68ccb1dc57a13d24b9db4dfe397c6c013ca7c7c22ee87","observation_id":"b5bc4c4e-73aa-49d1-bfac-37998a2d099f","resolution":{"observed_at":"2026-08-07T15:15:48.725718Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:50.637964Z","title":"A theoretical understanding of self-correction through in-context alignment","venue":null,"work_id":"d9ea38a5-2bde-4bdc-8ea6-5c3b8370226d","year":2024},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.730652Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:4ae2fdf781e97c3000f367a2088fba69736769400a4ad16afb638bd1b7fd9f8e","observation_id":"07420015-32fa-4fbc-a748-10cfb72eab27","resolution":{"observed_at":"2026-08-07T15:15:50.669636Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.16216","last_updated":"2026-05-16T18:30:31Z","snapshot_observed_at":"2026-08-12T16:47:35.127541Z","submitted_at":"2024-07-23T06:45:52Z","title":"Reinforcement Learning for LLM Post-Training: A Survey","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.16216","snapshot_observed_at":"2026-08-07T15:15:48.734701Z","title":"A comprehensive survey of llm alignment techniques: Rlhf, rlaif, ppo, dpo and more","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.734701Z"},"links":{"cited_paper":"/paper/2407.16216","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:c9bc008eb75d799d277153e561e48bfe8c162e294ca280714525c136f9149652","observation_id":"ed33eb3c-588d-40f4-b5d4-4f68de34dea2","resolution":{"observed_at":"2026-08-07T15:15:48.734701Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:50.520304Z","title":"Jailbroken: How does llm safety training fail? In NeurIPS, 2023","venue":null,"work_id":"86eb27e1-ee0a-498e-864d-af3ef6ae0f0a","year":2023},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.740043Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:12fda2a9dc51dce589067bb677750fc161731ad8154c3cde41f287b0b1b76cda","observation_id":"bedce304-5b00-4bfa-a80b-78b18299dbf8","resolution":{"observed_at":"2026-08-07T15:15:50.569860Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06387","last_updated":"2024-05-25T07:01:15Z","snapshot_observed_at":"2026-08-13T07:21:48.498707Z","submitted_at":"2023-10-10T07:50:29Z","title":"Jailbreak and Guard Aligned Language Models with Only Few In-Context Demonstrations","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06387","snapshot_observed_at":"2026-08-07T15:15:48.745729Z","title":"Jailbreak and guard aligned language models with only few in-context demonstrations","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.745729Z"},"links":{"cited_paper":"/paper/2310.06387","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:9445d3a17b38c95f4d443895f7e60eebaaa9744d436eb94addf0fa5e501c7f7d","observation_id":"9bdd8198-b371-4ea6-b84c-bdc93100f652","resolution":{"observed_at":"2026-08-07T15:15:48.745729Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:48.750950Z","title":"Certifiably robust rag against retrieval corruption","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.750950Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:30eceb6c9d36c3f6f88e0f3ac1995f6c57c2a1de51a55e34b7086c7ed5aa343b","observation_id":"80923eb6-3302-4e01-aefa-cd38a4aa9859","resolution":{"observed_at":"2026-08-07T15:15:48.750950Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:48.756297Z","title":"Defending chatgpt against jailbreak attack via self-reminders","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.756297Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:aec996df5d33ef42999227205080d8b1f2ee2ec9637b50df52a41ca5ce3af6b2","observation_id":"f959c3cd-4489-4b6b-a410-054ac506d400","resolution":{"observed_at":"2026-08-07T15:15:48.756297Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:50.395564Z","title":"Safedecoding: Defending against jailbreak attacks via safety-aware decoding","venue":null,"work_id":"c9cabe6d-874d-43bf-95cb-8e06b9b7a5f9","year":2024},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.760766Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:c6a74275f1f7597efc9a9c3946ac75f1f52f474911f7d56f6c80f599adf2bb9b","observation_id":"3f2115bb-ba62-431e-b74f-8fc169eb8c44","resolution":{"observed_at":"2026-08-07T15:15:50.435836Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.00083","last_updated":"2024-06-06T13:38:42Z","snapshot_observed_at":"2026-08-12T23:51:59.362331Z","submitted_at":"2024-06-03T02:25:33Z","title":"BadRAG: Identifying Vulnerabilities in Retrieval Augmented Generation of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.00083","snapshot_observed_at":"2026-08-07T15:15:48.764754Z","title":"Badrag: Identifying vulnerabilities in retrieval augmented generation of large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.764754Z"},"links":{"cited_paper":"/paper/2406.00083","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:b0d6d2e06d28bc02fa9b941c1349d7e734f82b877d913a93add62b10850b4190","observation_id":"d7afac52-f6a5-4c8d-ab70-f10d27949b6a","resolution":{"observed_at":"2026-08-07T15:15:48.764754Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:50.148540Z","title":"GPT-4 is too smart to be safe: Stealthy chat with LLMs via cipher","venue":null,"work_id":"9231d51c-3f49-4d09-9208-8d5a8ee8f90b","year":2024},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.770152Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:ac2661adab782b3cc0cd4eb39a487844a240a503fccf89d8b39f1c8bd213b29c","observation_id":"d2e0e766-d937-4532-ba55-fbcc41433d8d","resolution":{"observed_at":"2026-08-07T15:15:50.261654Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:48.774898Z","title":"The ai alignment problem: why it is hard, and where to start","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.774898Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:85c0d013d0010c94de90d5ccb1ee71f2d00e82f82c6736b08651231dc76ff0ca","observation_id":"98157b43-ce31-4c47-8cb9-741cf18ad1b1","resolution":{"observed_at":"2026-08-07T15:15:48.774898Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:49.802977Z","title":"How johnny can persuade llms to jailbreak them: Rethinking persuasion to challenge ai safety by humanizing llms","venue":null,"work_id":"70c7bd45-8be8-4ad3-be36-8a412bfc40ab","year":2024},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.779396Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:8de621464dcb7017e1a80407e03d51899b4e3a4b011a426ce5b565553cac1adc","observation_id":"4648f707-0efa-4133-9d6f-04b7ed3ff604","resolution":{"observed_at":"2026-08-07T15:15:50.030436Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:49.642005Z","title":"Boosting jailbreak attack with momentum","venue":null,"work_id":"1923d1b9-fad1-4173-9cc1-a3c90cc27b31","year":2025},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.785059Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:50d68bb3cd9859262165161710ce52a57bf2ade8dda46ee0dc36bee261281cd4","observation_id":"80e84f86-572f-43b4-adf7-e09f88940881","resolution":{"observed_at":"2026-08-07T15:15:49.714847Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.14924","last_updated":"2024-09-23T11:20:20Z","snapshot_observed_at":"2026-08-13T05:59:29.635188Z","submitted_at":"2024-09-23T11:20:20Z","title":"Retrieval Augmented Generation (RAG) and Beyond: A Comprehensive Survey on How to Make your LLMs use External Data More Wisely","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.14924","snapshot_observed_at":"2026-08-07T15:15:48.790118Z","title":"Retrieval augmented generation (rag) and beyond: A comprehensive survey on how to make your llms use external data more wisely","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.790118Z"},"links":{"cited_paper":"/paper/2409.14924","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:ee897fac6c31d63727407d567d4b8a520c89d591cb093e08db2d4b8e5590e458","observation_id":"252c035a-fe45-41e2-80c2-4547273f67ef","resolution":{"observed_at":"2026-08-07T15:15:48.790118Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:15:49.375614Z","title":"On prompt-driven safeguarding for large language models","venue":null,"work_id":"eb266093-cdf9-4a39-ac21-b4d33ebb8264","year":2024},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.796220Z"},"links":{"citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:948bb1c6a7b8df8b616df2b28af3e23d4c2a7b6b7dd723140e622660f10837a9","observation_id":"fde56ada-4e21-4d97-9e75-fc1a3582ac16","resolution":{"observed_at":"2026-08-07T15:15:49.528671Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.19156","last_updated":"2023-10-29T21:13:31Z","snapshot_observed_at":"2026-08-13T05:38:00.587615Z","submitted_at":"2023-10-29T21:13:31Z","title":"Poisoning Retrieval Corpora by Injecting Adversarial Passages","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.19156","snapshot_observed_at":"2026-08-07T15:15:48.802863Z","title":"Poisoning retrieval corpora by injecting adversarial passages","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.802863Z"},"links":{"cited_paper":"/paper/2310.19156","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:60671d6eeaa13cba566402aac51a5494e479e7477d4d198bee7a3dadae11e07e","observation_id":"dee30cea-29b1-40b1-b489-8b9ad2672dad","resolution":{"observed_at":"2026-08-07T15:15:48.802863Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.10102","last_updated":"2026-05-16T07:15:07Z","snapshot_observed_at":"2026-08-04T19:12:49.508911Z","submitted_at":"2024-09-16T09:06:44Z","title":"Trustworthiness in Retrieval-Augmented Generation Systems: A Survey","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.10102","snapshot_observed_at":"2026-08-07T15:15:48.809235Z","title":"Trustworthiness in retrieval-augmented generation systems: A survey","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.809235Z"},"links":{"cited_paper":"/paper/2409.10102","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:ddfb038e2c542f93e2447f6b2afcf0f8a0f79770ffe085de07b8b82af0917d9e","observation_id":"e4007808-2e3a-4b2b-9c68-e66246ff8219","resolution":{"observed_at":"2026-08-07T15:15:48.809235Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.18111","last_updated":"2024-10-08T06:37:03Z","snapshot_observed_at":"2026-08-12T23:56:14.183479Z","submitted_at":"2024-05-28T12:18:50Z","title":"ATM: Adversarial Tuning Multi-agent System Makes a Robust Retrieval-Augmented Generator","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.18111","snapshot_observed_at":"2026-08-07T15:15:48.816711Z","title":"Atm: Adversarial tuning multi- agent system makes a robust retrieval-augmented generator","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.816711Z"},"links":{"cited_paper":"/paper/2405.18111","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:a79bb53b4ddfe42592fc1ffffbc580954ffe75b495d6c3fd1515eb7d907be4ef","observation_id":"94337f36-ae35-4de6-9ad1-7e7532814868","resolution":{"observed_at":"2026-08-07T15:15:48.816711Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15043","last_updated":"2023-12-20T20:48:57Z","snapshot_observed_at":"2026-08-12T09:06:50.363435Z","submitted_at":"2023-07-27T17:49:12Z","title":"Universal and Transferable Adversarial Attacks on Aligned Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15043","snapshot_observed_at":"2026-08-07T15:15:48.822543Z","title":"Universal and transferable adversarial attacks on aligned language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.822543Z"},"links":{"cited_paper":"/paper/2307.15043","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:c7326756e6dcb83943b00b1d85e852c9b0751c610ee47ee639dea51e06b5534e","observation_id":"e7d43d4a-9670-4242-8b07-6e026f2d3d14","resolution":{"observed_at":"2026-08-07T15:15:48.822543Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07867","last_updated":"2024-08-13T01:55:06Z","snapshot_observed_at":"2026-08-13T04:20:22.513606Z","submitted_at":"2024-02-12T18:28:36Z","title":"PoisonedRAG: Knowledge Corruption Attacks to Retrieval-Augmented Generation of Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07867","snapshot_observed_at":"2026-08-07T15:15:48.828674Z","title":"yes\" or","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:48.828674Z"},"links":{"cited_paper":"/paper/2402.07867","citing_paper":"/paper/2505.15753"},"observation_digest":"sha256:fc18eba9f0f76134933d721ce333474001ae47e7d5f78c1a8beff53bde239169","observation_id":"26e778b4-da45-4896-8dfd-0f438f6e2eee","resolution":{"observed_at":"2026-08-07T15:15:48.828674Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.15753","last_updated":"2025-05-21T16:58:14Z","latest_version":1,"primary_category":"cs.CR","snapshot_observed_at":"2026-08-13T13:33:25.980248Z","submitted_at":"2025-05-21T16:58:14Z","title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval"},"reference_resolution":{"displayed":58,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":42,"verified_exact":0,"verified_fuzzy":16},"total_outbound_references":58},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 58 of 58 outbound references and 3 inbound Pith citation observations for arXiv:2505.15753."}