{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TOHG7SUAXCKU2CASJ6ITWQIS5R","short_pith_number":"pith:TOHG7SUA","schema_version":"1.0","canonical_sha256":"9b8e6fca80b8954d08124f913b4112ec6cb676dcd50ab7be90ab7d42ba2c612d","source":{"kind":"arxiv","id":"2402.13926","version":1},"attestation_state":"computed","paper":{"title":"Large Language Models are Vulnerable to Bait-and-Switch Attacks for Generating Harmful Content","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Federico Bianchi, James Zou","submitted_at":"2024-02-21T16:46:36Z","abstract_excerpt":"The risks derived from large language models (LLMs) generating deceptive and damaging content have been the subject of considerable research, but even safe generations can lead to problematic downstream impacts. In our study, we shift the focus to how even safe text coming from LLMs can be easily turned into potentially dangerous content through Bait-and-Switch attacks. In such attacks, the user first prompts LLMs with safe questions and then employs a simple find-and-replace post-hoc technique to manipulate the outputs into harmful narratives. The alarming efficacy of this approach in generat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.13926","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-02-21T16:46:36Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ca4d92d7a2be35ac6c4ef989e022e4b1a7519e408450d1cbba8ac9539896d5c5","abstract_canon_sha256":"928ec863cf2c4d002a0de539f724daab4448e98e0e5c905f3afafbf0d0bf6e77"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:47:50.359312Z","signature_b64":"E0w25Lw7rrOJ4seNHhLzVu6heC5vKHwsN+FXjErcGR+z9/VSOVStj/1ZaJOKzSKiFcMyvCqDBiQ55USgJjffAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9b8e6fca80b8954d08124f913b4112ec6cb676dcd50ab7be90ab7d42ba2c612d","last_reissued_at":"2026-07-05T07:47:50.358621Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:47:50.358621Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large Language Models are Vulnerable to Bait-and-Switch Attacks for Generating Harmful Content","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Federico Bianchi, James Zou","submitted_at":"2024-02-21T16:46:36Z","abstract_excerpt":"The risks derived from large language models (LLMs) generating deceptive and damaging content have been the subject of considerable research, but even safe generations can lead to problematic downstream impacts. In our study, we shift the focus to how even safe text coming from LLMs can be easily turned into potentially dangerous content through Bait-and-Switch attacks. In such attacks, the user first prompts LLMs with safe questions and then employs a simple find-and-replace post-hoc technique to manipulate the outputs into harmful narratives. The alarming efficacy of this approach in generat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.13926","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.13926/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.13926","created_at":"2026-07-05T07:47:50.358729+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.13926v1","created_at":"2026-07-05T07:47:50.358729+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.13926","created_at":"2026-07-05T07:47:50.358729+00:00"},{"alias_kind":"pith_short_12","alias_value":"TOHG7SUAXCKU","created_at":"2026-07-05T07:47:50.358729+00:00"},{"alias_kind":"pith_short_16","alias_value":"TOHG7SUAXCKU2CAS","created_at":"2026-07-05T07:47:50.358729+00:00"},{"alias_kind":"pith_short_8","alias_value":"TOHG7SUA","created_at":"2026-07-05T07:47:50.358729+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.04136","citing_title":"A Technical Survey of Reinforcement Learning Techniques for Large Language Models","ref_index":5,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TOHG7SUAXCKU2CASJ6ITWQIS5R","json":"https://pith.science/pith/TOHG7SUAXCKU2CASJ6ITWQIS5R.json","graph_json":"https://pith.science/api/pith-number/TOHG7SUAXCKU2CASJ6ITWQIS5R/graph.json","events_json":"https://pith.science/api/pith-number/TOHG7SUAXCKU2CASJ6ITWQIS5R/events.json","paper":"https://pith.science/paper/TOHG7SUA"},"agent_actions":{"view_html":"https://pith.science/pith/TOHG7SUAXCKU2CASJ6ITWQIS5R","download_json":"https://pith.science/pith/TOHG7SUAXCKU2CASJ6ITWQIS5R.json","view_paper":"https://pith.science/paper/TOHG7SUA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.13926&json=true","fetch_graph":"https://pith.science/api/pith-number/TOHG7SUAXCKU2CASJ6ITWQIS5R/graph.json","fetch_events":"https://pith.science/api/pith-number/TOHG7SUAXCKU2CASJ6ITWQIS5R/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TOHG7SUAXCKU2CASJ6ITWQIS5R/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TOHG7SUAXCKU2CASJ6ITWQIS5R/action/storage_attestation","attest_author":"https://pith.science/pith/TOHG7SUAXCKU2CASJ6ITWQIS5R/action/author_attestation","sign_citation":"https://pith.science/pith/TOHG7SUAXCKU2CASJ6ITWQIS5R/action/citation_signature","submit_replication":"https://pith.science/pith/TOHG7SUAXCKU2CASJ6ITWQIS5R/action/replication_record"}},"created_at":"2026-07-05T07:47:50.358729+00:00","updated_at":"2026-07-05T07:47:50.358729+00:00"}