{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BRWQLE6Y6SI5H7ZG6CHFSVLHT4","short_pith_number":"pith:BRWQLE6Y","schema_version":"1.0","canonical_sha256":"0c6d0593d8f491d3ff26f08e5955679f31052b8dd444b2119df4e8eb967c062c","source":{"kind":"arxiv","id":"2410.05295","version":4},"attestation_state":"computed","paper":{"title":"AutoDAN-Turbo: A Lifelong Agent for Strategy Self-Exploration to Jailbreak LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CR","authors_text":"Bo Li, Chaowei Xiao, Edward Suh, Huan Sun, Patrick McDaniel, Peiran Li, Somesh Jha, Xiaogeng Liu, Yevgeniy Vorobeychik, Zhuoqing Mao","submitted_at":"2024-10-03T17:59:01Z","abstract_excerpt":"In this paper, we propose AutoDAN-Turbo, a black-box jailbreak method that can automatically discover as many jailbreak strategies as possible from scratch, without any human intervention or predefined scopes (e.g., specified candidate strategies), and use them for red-teaming. As a result, AutoDAN-Turbo can significantly outperform baseline methods, achieving a 74.3% higher average attack success rate on public benchmarks. Notably, AutoDAN-Turbo achieves an 88.5 attack success rate on GPT-4-1106-turbo. In addition, AutoDAN-Turbo is a unified framework that can incorporate existing human-desig"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.05295","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2024-10-03T17:59:01Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"884dce29f9a77285ea1026fcc5c8cdff57a3ad55033974f068f7d2d49c43780d","abstract_canon_sha256":"756ed194c3868e1d57bf488e3daea1e4626710d5dd6518ed0238ff9765224c99"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:52:14.959644Z","signature_b64":"3pxu7RvxK8YbDbKSY224nc5tnm1mZR54u4UAYaF8PNp5crqr4Fw1lgMjl1/4Ip/F5gEjBedmT6c1ZHnXjTeODw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0c6d0593d8f491d3ff26f08e5955679f31052b8dd444b2119df4e8eb967c062c","last_reissued_at":"2026-07-05T10:52:14.959150Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:52:14.959150Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AutoDAN-Turbo: A Lifelong Agent for Strategy Self-Exploration to Jailbreak LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CR","authors_text":"Bo Li, Chaowei Xiao, Edward Suh, Huan Sun, Patrick McDaniel, Peiran Li, Somesh Jha, Xiaogeng Liu, Yevgeniy Vorobeychik, Zhuoqing Mao","submitted_at":"2024-10-03T17:59:01Z","abstract_excerpt":"In this paper, we propose AutoDAN-Turbo, a black-box jailbreak method that can automatically discover as many jailbreak strategies as possible from scratch, without any human intervention or predefined scopes (e.g., specified candidate strategies), and use them for red-teaming. As a result, AutoDAN-Turbo can significantly outperform baseline methods, achieving a 74.3% higher average attack success rate on public benchmarks. Notably, AutoDAN-Turbo achieves an 88.5 attack success rate on GPT-4-1106-turbo. In addition, AutoDAN-Turbo is a unified framework that can incorporate existing human-desig"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.05295","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.05295/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.05295","created_at":"2026-07-05T10:52:14.959203+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.05295v4","created_at":"2026-07-05T10:52:14.959203+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.05295","created_at":"2026-07-05T10:52:14.959203+00:00"},{"alias_kind":"pith_short_12","alias_value":"BRWQLE6Y6SI5","created_at":"2026-07-05T10:52:14.959203+00:00"},{"alias_kind":"pith_short_16","alias_value":"BRWQLE6Y6SI5H7ZG","created_at":"2026-07-05T10:52:14.959203+00:00"},{"alias_kind":"pith_short_8","alias_value":"BRWQLE6Y","created_at":"2026-07-05T10:52:14.959203+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":24,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24166","citing_title":"Distributed Quality-Diversity Search for Toxicity in Large Language Models","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19887","citing_title":"FinRED: An Expert-Guided Benchmark Generation and Evaluation Framework for Financial LLM Red-Teaming","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06797","citing_title":"Korean Culture into LLM Alignment: Toward Cultural Coherence","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03647","citing_title":"Black-box, Adaptive, Efficient, Transferable, Harmful, Applicable... Attacks Are All You Need to Break LLMs","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11671","citing_title":"Runtime Skill Audit: Targeted Runtime Probing for Agent Skill Security","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21674","citing_title":"Adversarial Reframing: A Framework for Targeted Generation in Language Models","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2511.12710","citing_title":"Evolve the Method, Not the Prompts: Evolutionary Synthesis of Jailbreak Attacks on LLMs","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21362","citing_title":"LASH: Adaptive Semantic Hybridization for Black-Box Jailbreaking of Large Language Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18868","citing_title":"DarkLLM: Learning Language-Driven Adversarial Attacks with Large Language Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17380","citing_title":"ADR: An Agentic Detection System for Enterprise Agentic AI Security","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2512.07761","citing_title":"TROJail: Trajectory-Level Optimization for Multi-Turn Large Language Model Jailbreaks with Process Rewards","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12813","citing_title":"REALISTA: Realistic Latent Adversarial Attacks that Elicit LLM Hallucinations","ref_index":185,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05682","citing_title":"PersonaTeaming: Supporting Persona-Driven Red-Teaming for Generative AI","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10582","citing_title":"Guaranteed Jailbreaking Defense via Disrupt-and-Rectify Smoothing","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05682","citing_title":"PersonaTeaming: Supporting Persona-Driven Red-Teaming for Generative AI","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05116","citing_title":"On the Hardness of Junking LLMs","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18976","citing_title":"STAR-Teaming: A Strategy-Response Multiplex Network Approach to Automated LLM Red Teaming","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18789","citing_title":"ARES: Adaptive Red-Teaming and End-to-End Repair of Policy-Reward System","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12616","citing_title":"Every Picture Tells a Dangerous Story: Memory-Augmented Multi-Agent Jailbreak Attacks on VLMs","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12817","citing_title":"Understanding and Improving Continuous Adversarial Training for LLMs via In-context Learning Theory","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12232","citing_title":"TEMPLATEFUZZ: Fine-Grained Chat Template Fuzzing for Jailbreaking and Red Teaming LLMs","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11309","citing_title":"The Salami Slicing Threat: Exploiting Cumulative Risks in LLM Systems","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17614","citing_title":"Characterizing Model-Native Skills","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02647","citing_title":"ContextualJailbreak: Evolutionary Red-Teaming via Simulated Conversational Priming","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BRWQLE6Y6SI5H7ZG6CHFSVLHT4","json":"https://pith.science/pith/BRWQLE6Y6SI5H7ZG6CHFSVLHT4.json","graph_json":"https://pith.science/api/pith-number/BRWQLE6Y6SI5H7ZG6CHFSVLHT4/graph.json","events_json":"https://pith.science/api/pith-number/BRWQLE6Y6SI5H7ZG6CHFSVLHT4/events.json","paper":"https://pith.science/paper/BRWQLE6Y"},"agent_actions":{"view_html":"https://pith.science/pith/BRWQLE6Y6SI5H7ZG6CHFSVLHT4","download_json":"https://pith.science/pith/BRWQLE6Y6SI5H7ZG6CHFSVLHT4.json","view_paper":"https://pith.science/paper/BRWQLE6Y","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.05295&json=true","fetch_graph":"https://pith.science/api/pith-number/BRWQLE6Y6SI5H7ZG6CHFSVLHT4/graph.json","fetch_events":"https://pith.science/api/pith-number/BRWQLE6Y6SI5H7ZG6CHFSVLHT4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BRWQLE6Y6SI5H7ZG6CHFSVLHT4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BRWQLE6Y6SI5H7ZG6CHFSVLHT4/action/storage_attestation","attest_author":"https://pith.science/pith/BRWQLE6Y6SI5H7ZG6CHFSVLHT4/action/author_attestation","sign_citation":"https://pith.science/pith/BRWQLE6Y6SI5H7ZG6CHFSVLHT4/action/citation_signature","submit_replication":"https://pith.science/pith/BRWQLE6Y6SI5H7ZG6CHFSVLHT4/action/replication_record"}},"created_at":"2026-07-05T10:52:14.959203+00:00","updated_at":"2026-07-05T10:52:14.959203+00:00"}