{"as_of":"2026-08-09T18:32:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:31e4444ee16a5e1dddd18ead62f166ee147260cfe2e9323930f044666c8253ce","coverage":[{"denominator":44,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":44,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T19:18:47.341663Z","state":"measured"},{"denominator":45,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":45,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T21:08:04.673125Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-05T21:08:06.388576Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"cited_work":{"arxiv_id":"2507.06043","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.06043","snapshot_observed_at":"2026-08-05T21:08:06.388576Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","venue":"cs.CR","work_id":"0af107df-715f-4776-955c-1c87769ce879","year":2025},"citing_paper":{"arxiv_id":"2508.09473","last_updated":"2025-08-13T04:05:28Z","snapshot_observed_at":"2026-08-08T12:54:38.707142Z","submitted_at":"2025-08-13T04:05:28Z","title":"NeuronTune: Fine-Grained Neuron Modulation for Balanced Safety-Utility Alignment in LLMs","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-05T21:08:04.673125Z"},"links":{"cited_paper":"/paper/2507.06043","citing_paper":"/paper/2508.09473"},"observation_digest":"sha256:b047b16d6be720cb98199b09a9d1fe728489615504ac80674613c82523aedef4","observation_id":"7c1330a7-e8ce-46f9-b491-83cbe88844a7","resolution":{"observed_at":"2026-08-05T21:08:06.426230Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.06043/citation-record","integrity":"/paper/2507.06043/integrity","json":"/paper/2507.06043/citation-record.json","paper":"/paper/2507.06043"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2204.05862","last_updated":"2022-04-12T15:02:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-12T15:02:38Z","title":"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.05862","snapshot_observed_at":"2026-08-06T19:18:42.488602Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:42.488602Z"},"links":{"cited_paper":"/paper/2204.05862","citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:94a4cd8ad506c2c45a3071f734a3da625e7e8675e1f942e248c9fb974f15c5d2","observation_id":"20ab7275-4286-4557-9065-dbc99fd0f870","resolution":{"observed_at":"2026-08-06T19:18:42.488602Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-06T19:18:42.524938Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:42.524938Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:55333e6c67b337bb82d38c4bfffaae197c17260240b6ba5058f23063a23e4633","observation_id":"dca1a89b-0b75-4bdf-9f69-555eb9834939","resolution":{"observed_at":"2026-08-06T19:18:42.524938Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:18:42.645786Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:42.645786Z"},"links":{"citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:8bb5eb77d23c8454d9dedf65eb12e62e5b63c0d1b30d226234deaad04187dc5d","observation_id":"05791a07-f465-41f7-bb98-f0b59ef38b68","resolution":{"observed_at":"2026-08-06T19:18:42.645786Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08419","last_updated":"2024-07-18T18:24:57Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-12T15:38:28Z","title":"Jailbreaking Black Box Large Language Models in Twenty Queries","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.08419","snapshot_observed_at":"2026-08-06T19:18:42.749367Z","title":"Pappas, and Eric Wong","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:42.749367Z"},"links":{"cited_paper":"/paper/2310.08419","citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:518dc804d5f160b642c317c9bf773f05a2c597adf2410f1d6ab839b83d7a87ea","observation_id":"de9f5339-3746-4348-96d9-378f28bdd6ab","resolution":{"observed_at":"2026-08-06T19:18:42.749367Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:18:42.849506Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:42.849506Z"},"links":{"citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:af02723acf314f8aa90d5b7a80179851bc0a7dcac492e37929b7016dc36a143d","observation_id":"f211f1dd-e80a-4b48-9029-1116493f6a2a","resolution":{"observed_at":"2026-08-06T19:18:42.849506Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05668","last_updated":"2025-05-26T12:56:04Z","snapshot_observed_at":"2026-08-08T01:20:38.655253Z","submitted_at":"2024-02-08T13:42:50Z","title":"JailbreakRadar: Comprehensive Assessment of Jailbreak Attacks Against LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05668","snapshot_observed_at":"2026-08-06T19:18:42.941390Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:42.941390Z"},"links":{"cited_paper":"/paper/2402.05668","citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:bf6d80efab6dc2c7330e8682a60ea40434eea4df315df35dd9c212c97fb53666","observation_id":"155f482f-eb0c-4b09-930b-c5236f1f9e56","resolution":{"observed_at":"2026-08-06T19:18:42.941390Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.03274","last_updated":"2024-12-02T08:53:40Z","snapshot_observed_at":"2026-08-02T08:56:14.285582Z","submitted_at":"2024-09-05T06:31:37Z","title":"Recent Advances in Attack and Defense Approaches of Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.03274","snapshot_observed_at":"2026-08-06T19:18:43.023171Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:43.023171Z"},"links":{"cited_paper":"/paper/2409.03274","citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:e3cd0a9213fa4e0178834f5bfdd672f4d52909505bd38c94c3fa86f45b64cca2","observation_id":"c9a0a10f-545e-408e-a876-ba2746afa4df","resolution":{"observed_at":"2026-08-06T19:18:43.023171Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-06T19:18:43.106007Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:43.106007Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:c4f974ba98676c95004ae4672dcf7f6989d3c5a53a82244685d8ddc4feb1670b","observation_id":"c2213e70-c872-4395-afc6-7cd63728a498","resolution":{"observed_at":"2026-08-06T19:18:43.106007Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:18:43.155123Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:43.155123Z"},"links":{"citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:6e72754431ab5bbfd3072000b15b36d309b15c06d04d484aaf2c9af01d44b63b","observation_id":"66a459e4-4b2b-4c9f-9f7c-cc657cb21a57","resolution":{"observed_at":"2026-08-06T19:18:43.155123Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-06T19:18:43.324924Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:43.324924Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:8fad99ed47bb527d01fe515c77a8f3815340f9fb4835e7f98eae1cb9bd16e556","observation_id":"90cbfe8c-bc9c-412d-934d-4bf66a9f80dd","resolution":{"observed_at":"2026-08-06T19:18:43.324924Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:18:49.725088Z","title":null,"venue":null,"work_id":"38b34a94-12ac-4c22-a160-85dbefef49e1","year":2024},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:43.494982Z"},"links":{"citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:c23ba410b57debb9901ef8d6fbad8eb8ca7ffb9d6418111a76979183fdc42877","observation_id":"2a34fef4-bc96-4448-b0cc-ee6154146cec","resolution":{"observed_at":"2026-08-06T19:18:49.761528Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:18:49.554879Z","title":"Cai, James Wexler, Fernanda B","venue":null,"work_id":"1554270a-15cd-426e-9a4a-bc318624ea78","year":2017},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:43.629083Z"},"links":{"citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:02ec6dcc18048dc95d21e46041f22ff6702ae07e8759080a5d96fb4a8236e6f1","observation_id":"14825689-1475-4b1b-92ab-944630b02ac0","resolution":{"observed_at":"2026-08-06T19:18:49.602687Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:18:49.341227Z","title":null,"venue":null,"work_id":"8665facb-1034-4637-a764-3890360ce4ed","year":2024},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:43.749206Z"},"links":{"citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:3d9195597835ba5daf8d6066f9392e1a2792e5c3cd3f8ef0bb3dd0d2683ad72d","observation_id":"2baaf5bd-1d05-4f98-852d-db0b98c6f7f7","resolution":{"observed_at":"2026-08-06T19:18:49.424805Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:18:49.125424Z","title":null,"venue":null,"work_id":"ace0500f-e520-448e-9587-e24fa7da56c5","year":2024},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:43.839587Z"},"links":{"citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:85d9002add299365436e06ad9e5550e9bc5e73debf1088586ef57b913251cea2","observation_id":"ee87e432-a3ea-48b7-8126-f346b8db9b00","resolution":{"observed_at":"2026-08-06T19:18:49.237439Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:18:48.938944Z","title":null,"venue":null,"work_id":"2aabb6f6-2efa-43da-8cc1-c8be171e552e","year":2025},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:43.957717Z"},"links":{"citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:1a6fd9c63a911779e8280e046994c2f813bf78d95ae1bfbd2e8218b484e4b7d6","observation_id":"f4e61275-a567-406e-9d36-772ed10f644d","resolution":{"observed_at":"2026-08-06T19:18:49.039356Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:18:44.110225Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:44.110225Z"},"links":{"citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:4bb1f5d0d6f8f394c80a1a89e7c49161ab1001bea22330c25093112b483b4f39","observation_id":"ffdae9da-ff4b-4453-b3b6-53c2e2961b41","resolution":{"observed_at":"2026-08-06T19:18:44.110225Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"3530.36650","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:18:48.248505Z","title":null,"venue":null,"work_id":"aef0b10f-8a45-4e4b-b82f-e864acc523bf","year":2024},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:44.315764Z"},"links":{"citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:50409007554a0b12a3b8bbbb3ea0308b23ddd8d8ec3c7f98ff84f639461cd238","observation_id":"60fcaa8e-59eb-4e78-9b0e-88e058a1dce9","resolution":{"observed_at":"2026-08-06T19:18:48.293208Z","resolver_source":"raw_fallback","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.13860","last_updated":"2024-03-10T13:58:08Z","snapshot_observed_at":"2026-07-06T15:31:18.144952Z","submitted_at":"2023-05-23T09:33:38Z","title":"Jailbreaking ChatGPT via Prompt Engineering: An Empirical Study","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.13860","snapshot_observed_at":"2026-08-06T19:18:44.434891Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:44.434891Z"},"links":{"cited_paper":"/paper/2305.13860","citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:51d12e6265339beb96cc2a974045195741e4073620fd7b2ef52e674e29d7fb6d","observation_id":"92ec32f8-01f8-42ac-9f86-3b9e0d7f35e5","resolution":{"observed_at":"2026-08-06T19:18:44.434891Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:18:48.755323Z","title":null,"venue":null,"work_id":"d19d9ef4-80e8-4b00-a9f0-1119b4a89872","year":2023},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:44.557865Z"},"links":{"citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:caa3db65892d207d38082af3a637deef2ea9fb1e4474ce93d4dc2d3d8b0994ea","observation_id":"e75817f1-eb76-4a4d-ad33-e6183d1af717","resolution":{"observed_at":"2026-08-06T19:18:48.877206Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-06T19:18:44.729046Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:44.729046Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:72338ca975d9884929a0c8e118f13a902b3dbf66178e1c6b6151c5d12b3c44d9","observation_id":"aa26c873-6fc4-4540-aa2b-924dd1d18413","resolution":{"observed_at":"2026-08-06T19:18:44.729046Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:18:44.823274Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:44.823274Z"},"links":{"citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:8eb87d2241cc47813d2cafba93125686d398b9bd2f9d6e2bc3e157e6afa3f5f3","observation_id":"a67b6306-0fed-4213-9137-12a769f00d0f","resolution":{"observed_at":"2026-08-06T19:18:44.823274Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-06T19:18:44.931351Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:44.931351Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:3782e8d5a65eb1bc560b16f10ddf65a4ed32cf3915abdcf2eac9af519d1c6cf1","observation_id":"79a238d2-a2d6-4640-99b8-806ff726846c","resolution":{"observed_at":"2026-08-06T19:18:44.931351Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.18290","last_updated":"2024-07-29T22:26:36Z","snapshot_observed_at":"2026-08-01T16:34:38.795326Z","submitted_at":"2023-05-29T17:57:46Z","title":"Direct Preference Optimization: Your Language Model is Secretly a Reward Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.18290","snapshot_observed_at":"2026-08-06T19:18:45.033818Z","title":"Manning, and Chelsea Finn","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:45.033818Z"},"links":{"cited_paper":"/paper/2305.18290","citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:268302ee62c8e5930559f26e2f4cc935409414b0147cffc8f17338f917fdf375","observation_id":"39ceda5e-54fd-49f1-883c-711e3adea093","resolution":{"observed_at":"2026-08-06T19:18:45.033818Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:18:45.148774Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:45.148774Z"},"links":{"citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:8d65ca0592ffaf85841897ef83cf153b2401ebcc64a4325adca3aef23342a4b1","observation_id":"f26780fa-9390-45b8-bf79-0ac77c907ffc","resolution":{"observed_at":"2026-08-06T19:18:45.148774Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03684","last_updated":"2024-06-11T19:02:52Z","snapshot_observed_at":"2026-07-06T16:28:22.350574Z","submitted_at":"2023-10-05T17:01:53Z","title":"SmoothLLM: Defending Large Language Models Against Jailbreaking Attacks","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03684","snapshot_observed_at":"2026-08-06T19:18:45.243985Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:45.243985Z"},"links":{"cited_paper":"/paper/2310.03684","citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:e377eb165eaadd23553477be7c706d01f88226a32634b56f905e491a0876cbf8","observation_id":"b578c107-79d4-4f86-b42f-35891a32855f","resolution":{"observed_at":"2026-08-06T19:18:45.243985Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.03825","last_updated":"2024-05-15T12:06:31Z","snapshot_observed_at":"2026-07-06T16:03:34.432602Z","submitted_at":"2023-08-07T16:55:20Z","title":"\"Do Anything Now\": Characterizing and Evaluating In-The-Wild Jailbreak Prompts on Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.03825","snapshot_observed_at":"2026-08-06T19:18:45.379143Z","title":"do anything now","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:45.379143Z"},"links":{"cited_paper":"/paper/2308.03825","citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:2866becd233d0b908a27ceef1fc4291ba19d61d74bf013a8ae260e4f83fb5a03","observation_id":"b1715f98-6308-4bfa-a4e3-663672797a83","resolution":{"observed_at":"2026-08-06T19:18:45.379143Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.10260","last_updated":"2024-08-27T03:32:47Z","snapshot_observed_at":"2026-08-04T21:32:35.483431Z","submitted_at":"2024-02-15T18:58:09Z","title":"A StrongREJECT for Empty Jailbreaks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.10260","snapshot_observed_at":"2026-08-06T19:18:45.500292Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:45.500292Z"},"links":{"cited_paper":"/paper/2402.10260","citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:66d1f66e37d77a41158c9358d288300482bf9e86a0e18347523204483a805aba","observation_id":"5284303c-4ce3-43b0-a30c-41b934161202","resolution":{"observed_at":"2026-08-06T19:18:45.500292Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:18:45.634496Z","title":"Hashimoto","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:45.634496Z"},"links":{"citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:de1577904d009452e190ac32404a1d899a8afaa1406ece3e47bda18fbcb7190b","observation_id":"8d8f141d-59c6-49ae-ba2b-348faa9e9ffc","resolution":{"observed_at":"2026-08-06T19:18:45.634496Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:18:45.728426Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:45.728426Z"},"links":{"citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:6e0959d997fa16c5b8165a499aa145b94254231bc876a186e13d733b70a1d34b","observation_id":"758c82ee-0cc8-4b74-a19b-d0aaa8d1712a","resolution":{"observed_at":"2026-08-06T19:18:45.728426Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02483","last_updated":"2023-07-05T17:58:10Z","snapshot_observed_at":"2026-08-08T18:11:45.725753Z","submitted_at":"2023-07-05T17:58:10Z","title":"Jailbroken: How Does LLM Safety Training Fail?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.02483","snapshot_observed_at":"2026-08-06T19:18:45.849258Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:45.849258Z"},"links":{"cited_paper":"/paper/2307.02483","citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:bd15c8fc33ee706666fa59df94327b2e7701101cfaedffeda0dfaeca3b310558","observation_id":"bce1bd3a-1c87-4860-93ae-a6b77a32d2df","resolution":{"observed_at":"2026-08-06T19:18:45.849258Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:18:45.952671Z","title":"Dai, and Quoc V Le","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:45.952671Z"},"links":{"citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:2343a89907300cbadbae6760aae7fb2a692deab09b99527daeb7efd8b24e10b9","observation_id":"0a191a1b-a55e-4e05-8ab0-ed4870e40c1e","resolution":{"observed_at":"2026-08-06T19:18:45.952671Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.04359","last_updated":"2021-12-08T16:09:48Z","snapshot_observed_at":"2026-08-09T15:17:43.394064Z","submitted_at":"2021-12-08T16:09:48Z","title":"Ethical and social risks of harm from Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.04359","snapshot_observed_at":"2026-08-06T19:18:46.047757Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:46.047757Z"},"links":{"cited_paper":"/paper/2112.04359","citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:9f2637e9fcfa2768001f001765031d23fc1f4a0e0a13ed991bbf95835627b1df","observation_id":"8f7b9bf5-41b1-4bcc-b490-cae65e440cd3","resolution":{"observed_at":"2026-08-06T19:18:46.047757Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:18:46.137678Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:46.137678Z"},"links":{"citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:08171b7eb230c6f375eddc348ee3b105534ba4e6de04ec24750d09502f93335b","observation_id":"dd2c71f7-cde5-4944-bc27-36a99e3ed40e","resolution":{"observed_at":"2026-08-06T19:18:46.137678Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:18:48.516367Z","title":null,"venue":null,"work_id":"ff40dfdb-aefe-4166-8f3b-db6fc8e6b587","year":2024},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:46.223411Z"},"links":{"citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:5b9d8c2aed5db7032618af329bba2a9f4a408502ad270d8fba7cd92875bbfe1d","observation_id":"65c38dfd-c68d-406b-834d-8f98bd11a601","resolution":{"observed_at":"2026-08-06T19:18:48.613526Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:18:46.295508Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:46.295508Z"},"links":{"citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:050ea71ebb20a66c702056cd1bf3615e19a724f5de76064f1dcd19375a38ddf2","observation_id":"2bca0902-16d6-4263-b795-1ee6f7ec13bb","resolution":{"observed_at":"2026-08-06T19:18:46.295508Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04295","last_updated":"2024-08-30T11:57:47Z","snapshot_observed_at":"2026-08-04T23:34:13.332065Z","submitted_at":"2024-07-05T06:57:30Z","title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04295","snapshot_observed_at":"2026-08-06T19:18:46.428000Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:46.428000Z"},"links":{"cited_paper":"/paper/2407.04295","citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:2dd422839f5b41adf059e6c917ed46bee3f82a73b45068b7eca5fedd59d9e082","observation_id":"84632e61-765a-4cf8-b5b5-9911eab0c50f","resolution":{"observed_at":"2026-08-06T19:18:46.428000Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02446","last_updated":"2024-01-27T22:54:52Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-03T21:30:56Z","title":"Low-Resource Languages Jailbreak GPT-4","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02446","snapshot_observed_at":"2026-08-06T19:18:46.549467Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:46.549467Z"},"links":{"cited_paper":"/paper/2310.02446","citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:28171b6f98b517de6bdd17327af56490864199973a63b75daa58eb5c2fccea5a","observation_id":"70640b84-e213-4601-b2c0-aa0b955aa74c","resolution":{"observed_at":"2026-08-06T19:18:46.549467Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.07343","last_updated":"2023-10-11T09:46:32Z","snapshot_observed_at":"2026-07-06T16:31:08.082416Z","submitted_at":"2023-10-11T09:46:32Z","title":"How Do Large Language Models Capture the Ever-changing World Knowledge? A Review of Recent Advances","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.07343","snapshot_observed_at":"2026-08-06T19:18:46.662354Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:46.662354Z"},"links":{"cited_paper":"/paper/2310.07343","citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:b52abc109b3575c82591b84649188d27db7e90f13f96d71a47fa5af19657fc4a","observation_id":"87b57c8e-c12e-49b7-a269-950a04fd0d57","resolution":{"observed_at":"2026-08-06T19:18:46.662354Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:18:46.786950Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:46.786950Z"},"links":{"citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:07a8d5cbf9a87049a71da48473e09fe27712e4f9f9089f3cac90928166178b9d","observation_id":"36385948-cc43-4a26-8b82-35896d8f38de","resolution":{"observed_at":"2026-08-06T19:18:46.786950Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:18:46.864156Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:46.864156Z"},"links":{"citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:7d34819d3a83d051412415e330812e172e1a0e49bd3e46f377840e4c0754f05c","observation_id":"66a5804c-728e-4412-a414-3aaff9b71d58","resolution":{"observed_at":"2026-08-06T19:18:46.864156Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01405","last_updated":"2025-03-03T06:14:14Z","snapshot_observed_at":"2026-07-06T16:26:38.284922Z","submitted_at":"2023-10-02T17:59:07Z","title":"Representation Engineering: A Top-Down Approach to AI Transparency","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01405","snapshot_observed_at":"2026-08-06T19:18:46.970519Z","title":"Byun, Zifan Wang, Alex Mallen, Steven Basart, Sanmi Koyejo, Dawn Song, Matt Fredrikson, J","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:46.970519Z"},"links":{"cited_paper":"/paper/2310.01405","citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:90f941fe335a10fe7fde5bd3d28b318d483297eb42376a9e00e9c168e04fe05f","observation_id":"0f3429be-0dbf-4ac4-8253-5162e656584a","resolution":{"observed_at":"2026-08-06T19:18:46.970519Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15043","last_updated":"2023-12-20T20:48:57Z","snapshot_observed_at":"2026-07-06T15:59:23.019044Z","submitted_at":"2023-07-27T17:49:12Z","title":"Universal and Transferable Adversarial Attacks on Aligned Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15043","snapshot_observed_at":"2026-08-06T19:18:47.094379Z","title":"Zico Kolter, and Matt Fredrikson","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:47.094379Z"},"links":{"cited_paper":"/paper/2307.15043","citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:048f489e0959988224bf2aeea2662d6aef6e4180db1c5503f9198cd3a45a2ce9","observation_id":"822a0e55-2997-4ae1-8622-df97a4992f59","resolution":{"observed_at":"2026-08-06T19:18:47.094379Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:18:47.235990Z","title":"online\" 'onlinestring :=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:47.235990Z"},"links":{"citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:42f2e444cf7cb913af26b52303cdc1dd3253be3b285290c8bd5216fc5a0e74e1","observation_id":"86dd1667-75f1-4985-ba75-b247edf29fbc","resolution":{"observed_at":"2026-08-06T19:18:47.235990Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:18:47.341663Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:47.341663Z"},"links":{"citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:a19a8ef2b230be5049920914da1aa2ffa79e6c8ff4ea438713dec26f06b34dc8","observation_id":"13d529b9-ac61-4a1e-8e94-43b2e793f351","resolution":{"observed_at":"2026-08-06T19:18:47.341663Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","latest_version":2,"primary_category":"cs.CR","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations"},"reference_resolution":{"displayed":44,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":42,"verified_exact":0,"verified_fuzzy":1},"total_outbound_references":44},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 44 of 44 outbound references and 1 inbound Pith citation observation for arXiv:2507.06043."}