{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LEWJBSI6KWF2RCTONG4YKX3WD6","short_pith_number":"pith:LEWJBSI6","schema_version":"1.0","canonical_sha256":"592c90c91e558ba88a6e69b9855f761f86ce23c60dd00248826b581b639ac173","source":{"kind":"arxiv","id":"2505.10066","version":1},"attestation_state":"computed","paper":{"title":"Dark LLMs: The Growing Threat of Unaligned AI Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Adi Wasenstein, Lior Rokach, Michael Fire, Yitzhak Elbazis","submitted_at":"2025-05-15T08:07:04Z","abstract_excerpt":"Large Language Models (LLMs) rapidly reshape modern life, advancing fields from healthcare to education and beyond. However, alongside their remarkable capabilities lies a significant threat: the susceptibility of these models to jailbreaking. The fundamental vulnerability of LLMs to jailbreak attacks stems from the very data they learn from. As long as this training data includes unfiltered, problematic, or 'dark' content, the models can inherently learn undesirable patterns or weaknesses that allow users to circumvent their intended safety controls. Our research identifies the growing threat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.10066","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-15T08:07:04Z","cross_cats_sorted":["cs.AI","cs.CR","cs.LG"],"title_canon_sha256":"872c80da07e59612050c3696507f97f0293289c04848c8ec5e9e90f6a8da5544","abstract_canon_sha256":"2de085e93541040ac16cd9e398cddb539c824da10abd9ce29fedded7a58560ca"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:03:32.700472Z","signature_b64":"8fE4Wx55oXiLveREtEk6yfp3Xk5G2s208mItd4xiJ8TPSBQOXjEvtEhr7LGqWk8/7FOnqxro1PKyowhAxQiRCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"592c90c91e558ba88a6e69b9855f761f86ce23c60dd00248826b581b639ac173","last_reissued_at":"2026-07-05T11:03:32.699984Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:03:32.699984Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dark LLMs: The Growing Threat of Unaligned AI Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Adi Wasenstein, Lior Rokach, Michael Fire, Yitzhak Elbazis","submitted_at":"2025-05-15T08:07:04Z","abstract_excerpt":"Large Language Models (LLMs) rapidly reshape modern life, advancing fields from healthcare to education and beyond. However, alongside their remarkable capabilities lies a significant threat: the susceptibility of these models to jailbreaking. The fundamental vulnerability of LLMs to jailbreak attacks stems from the very data they learn from. As long as this training data includes unfiltered, problematic, or 'dark' content, the models can inherently learn undesirable patterns or weaknesses that allow users to circumvent their intended safety controls. Our research identifies the growing threat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.10066","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.10066/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.10066","created_at":"2026-07-05T11:03:32.700043+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.10066v1","created_at":"2026-07-05T11:03:32.700043+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.10066","created_at":"2026-07-05T11:03:32.700043+00:00"},{"alias_kind":"pith_short_12","alias_value":"LEWJBSI6KWF2","created_at":"2026-07-05T11:03:32.700043+00:00"},{"alias_kind":"pith_short_16","alias_value":"LEWJBSI6KWF2RCTO","created_at":"2026-07-05T11:03:32.700043+00:00"},{"alias_kind":"pith_short_8","alias_value":"LEWJBSI6","created_at":"2026-07-05T11:03:32.700043+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28526","citing_title":"A French OSCE Dialogue Dataset and Controllable Virtual Patient System for Clinical Training","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11772","citing_title":"Towards Automated Pentesting with Large Language Models","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LEWJBSI6KWF2RCTONG4YKX3WD6","json":"https://pith.science/pith/LEWJBSI6KWF2RCTONG4YKX3WD6.json","graph_json":"https://pith.science/api/pith-number/LEWJBSI6KWF2RCTONG4YKX3WD6/graph.json","events_json":"https://pith.science/api/pith-number/LEWJBSI6KWF2RCTONG4YKX3WD6/events.json","paper":"https://pith.science/paper/LEWJBSI6"},"agent_actions":{"view_html":"https://pith.science/pith/LEWJBSI6KWF2RCTONG4YKX3WD6","download_json":"https://pith.science/pith/LEWJBSI6KWF2RCTONG4YKX3WD6.json","view_paper":"https://pith.science/paper/LEWJBSI6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.10066&json=true","fetch_graph":"https://pith.science/api/pith-number/LEWJBSI6KWF2RCTONG4YKX3WD6/graph.json","fetch_events":"https://pith.science/api/pith-number/LEWJBSI6KWF2RCTONG4YKX3WD6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LEWJBSI6KWF2RCTONG4YKX3WD6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LEWJBSI6KWF2RCTONG4YKX3WD6/action/storage_attestation","attest_author":"https://pith.science/pith/LEWJBSI6KWF2RCTONG4YKX3WD6/action/author_attestation","sign_citation":"https://pith.science/pith/LEWJBSI6KWF2RCTONG4YKX3WD6/action/citation_signature","submit_replication":"https://pith.science/pith/LEWJBSI6KWF2RCTONG4YKX3WD6/action/replication_record"}},"created_at":"2026-07-05T11:03:32.700043+00:00","updated_at":"2026-07-05T11:03:32.700043+00:00"}