{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2MYBB4OQL3MDLO2MQHW2FL7QVO","short_pith_number":"pith:2MYBB4OQ","schema_version":"1.0","canonical_sha256":"d33010f1d05ed835bb4c81eda2aff0ab9d5a3159d624a3f59ffaae7592680d8a","source":{"kind":"arxiv","id":"2402.19464","version":1},"attestation_state":"computed","paper":{"title":"Curiosity-driven Red-teaming for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Akash Srivastava, Aldo Pareja, Idan Shenfeld, James Glass, Pulkit Agrawal, Tsun-Hsuan Wang, Yung-Sung Chuang, Zhang-Wei Hong","submitted_at":"2024-02-29T18:55:03Z","abstract_excerpt":"Large language models (LLMs) hold great potential for many natural language applications but risk generating incorrect or toxic content. To probe when an LLM generates unwanted content, the current paradigm is to recruit a \\textit{red team} of human testers to design input prompts (i.e., test cases) that elicit undesirable responses from LLMs. However, relying solely on human testers is expensive and time-consuming. Recent works automate red teaming by training a separate red team LLM with reinforcement learning (RL) to generate test cases that maximize the chance of eliciting undesirable resp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.19464","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-29T18:55:03Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"a45fd6248974ad03b205a92bc88dacdea7109a107615fd5984888137e4c20952","abstract_canon_sha256":"0acf2489db72508ce5c0e32f4b5e06bf14ac5c493b17860bdce07b4e808a623a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:50:45.653275Z","signature_b64":"SKWLVywczE2+JoEGvS73C2sCH920Y+c7gD8rI1sObLQR9RqimuVRqRqysAsyGE+93sL+iQs2vvgva1dbPiF6CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d33010f1d05ed835bb4c81eda2aff0ab9d5a3159d624a3f59ffaae7592680d8a","last_reissued_at":"2026-07-05T07:50:45.652769Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:50:45.652769Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Curiosity-driven Red-teaming for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Akash Srivastava, Aldo Pareja, Idan Shenfeld, James Glass, Pulkit Agrawal, Tsun-Hsuan Wang, Yung-Sung Chuang, Zhang-Wei Hong","submitted_at":"2024-02-29T18:55:03Z","abstract_excerpt":"Large language models (LLMs) hold great potential for many natural language applications but risk generating incorrect or toxic content. To probe when an LLM generates unwanted content, the current paradigm is to recruit a \\textit{red team} of human testers to design input prompts (i.e., test cases) that elicit undesirable responses from LLMs. However, relying solely on human testers is expensive and time-consuming. Recent works automate red teaming by training a separate red team LLM with reinforcement learning (RL) to generate test cases that maximize the chance of eliciting undesirable resp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.19464","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.19464/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.19464","created_at":"2026-07-05T07:50:45.652824+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.19464v1","created_at":"2026-07-05T07:50:45.652824+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.19464","created_at":"2026-07-05T07:50:45.652824+00:00"},{"alias_kind":"pith_short_12","alias_value":"2MYBB4OQL3MD","created_at":"2026-07-05T07:50:45.652824+00:00"},{"alias_kind":"pith_short_16","alias_value":"2MYBB4OQL3MDLO2M","created_at":"2026-07-05T07:50:45.652824+00:00"},{"alias_kind":"pith_short_8","alias_value":"2MYBB4OQ","created_at":"2026-07-05T07:50:45.652824+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2412.02818","citing_title":"RoboMD: Uncovering Robot Vulnerabilities through Semantic Potential Fields","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2510.07239","citing_title":"Red-Bandit: Test-Time Adaptation for LLM Red-Teaming via Bandit-Guided LoRA Experts","ref_index":48,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2MYBB4OQL3MDLO2MQHW2FL7QVO","json":"https://pith.science/pith/2MYBB4OQL3MDLO2MQHW2FL7QVO.json","graph_json":"https://pith.science/api/pith-number/2MYBB4OQL3MDLO2MQHW2FL7QVO/graph.json","events_json":"https://pith.science/api/pith-number/2MYBB4OQL3MDLO2MQHW2FL7QVO/events.json","paper":"https://pith.science/paper/2MYBB4OQ"},"agent_actions":{"view_html":"https://pith.science/pith/2MYBB4OQL3MDLO2MQHW2FL7QVO","download_json":"https://pith.science/pith/2MYBB4OQL3MDLO2MQHW2FL7QVO.json","view_paper":"https://pith.science/paper/2MYBB4OQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.19464&json=true","fetch_graph":"https://pith.science/api/pith-number/2MYBB4OQL3MDLO2MQHW2FL7QVO/graph.json","fetch_events":"https://pith.science/api/pith-number/2MYBB4OQL3MDLO2MQHW2FL7QVO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2MYBB4OQL3MDLO2MQHW2FL7QVO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2MYBB4OQL3MDLO2MQHW2FL7QVO/action/storage_attestation","attest_author":"https://pith.science/pith/2MYBB4OQL3MDLO2MQHW2FL7QVO/action/author_attestation","sign_citation":"https://pith.science/pith/2MYBB4OQL3MDLO2MQHW2FL7QVO/action/citation_signature","submit_replication":"https://pith.science/pith/2MYBB4OQL3MDLO2MQHW2FL7QVO/action/replication_record"}},"created_at":"2026-07-05T07:50:45.652824+00:00","updated_at":"2026-07-05T07:50:45.652824+00:00"}