{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:EPX5XKZFGRDZB4HP4OT7KDRPZG","short_pith_number":"pith:EPX5XKZF","schema_version":"1.0","canonical_sha256":"23efdbab25344790f0efe3a7f50e2fc9b986b2802a03e15afafad15024d7088a","source":{"kind":"arxiv","id":"2505.23793","version":1},"attestation_state":"computed","paper":{"title":"USB: A Comprehensive and Unified Safety Evaluation Benchmark for Multimodal Large Language Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Baolin Zheng, Bo Zheng, Guanlin Chen, Hongqiong Zhong, Huiyun Jing, Jiaheng Liu, Jian Yang, Jincheng Wei, Kaifu Zhang, Qingyang Teng, Weixun Wang, Wenbo Su, Xiaoyong Zhu, Yingshui Tan, Zhendong Liu","submitted_at":"2025-05-26T08:39:14Z","abstract_excerpt":"Despite their remarkable achievements and widespread adoption, Multimodal Large Language Models (MLLMs) have revealed significant security vulnerabilities, highlighting the urgent need for robust safety evaluation benchmarks. Existing MLLM safety benchmarks, however, fall short in terms of data quality and coverge, and modal risk combinations, resulting in inflated and contradictory evaluation results, which hinders the discovery and governance of security concerns. Besides, we argue that vulnerabilities to harmful queries and oversensitivity to harmless ones should be considered simultaneousl"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.23793","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CR","submitted_at":"2025-05-26T08:39:14Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"bce7ff04bd546fe8a08b0ee4ec29bd2b91ebb6ec97d838dc6c1711c4649b35b5","abstract_canon_sha256":"ac27bd08d2d56d78f3738ee6d9a77283c73be4f93e07c52bf2aedafe40807229"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:12:35.558364Z","signature_b64":"578+91iqNjHmV8zLoUMVGQTFAIMXDXKY0M5S4+Q11LivX2ipxIGGtofSrqcZh3NS8OIe9y2AHK6Cmwr2T+0BDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"23efdbab25344790f0efe3a7f50e2fc9b986b2802a03e15afafad15024d7088a","last_reissued_at":"2026-07-05T11:12:35.557755Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:12:35.557755Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"USB: A Comprehensive and Unified Safety Evaluation Benchmark for Multimodal Large Language Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Baolin Zheng, Bo Zheng, Guanlin Chen, Hongqiong Zhong, Huiyun Jing, Jiaheng Liu, Jian Yang, Jincheng Wei, Kaifu Zhang, Qingyang Teng, Weixun Wang, Wenbo Su, Xiaoyong Zhu, Yingshui Tan, Zhendong Liu","submitted_at":"2025-05-26T08:39:14Z","abstract_excerpt":"Despite their remarkable achievements and widespread adoption, Multimodal Large Language Models (MLLMs) have revealed significant security vulnerabilities, highlighting the urgent need for robust safety evaluation benchmarks. Existing MLLM safety benchmarks, however, fall short in terms of data quality and coverge, and modal risk combinations, resulting in inflated and contradictory evaluation results, which hinders the discovery and governance of security concerns. Besides, we argue that vulnerabilities to harmful queries and oversensitivity to harmless ones should be considered simultaneousl"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.23793","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.23793/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.23793","created_at":"2026-07-05T11:12:35.557825+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.23793v1","created_at":"2026-07-05T11:12:35.557825+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.23793","created_at":"2026-07-05T11:12:35.557825+00:00"},{"alias_kind":"pith_short_12","alias_value":"EPX5XKZFGRDZ","created_at":"2026-07-05T11:12:35.557825+00:00"},{"alias_kind":"pith_short_16","alias_value":"EPX5XKZFGRDZB4HP","created_at":"2026-07-05T11:12:35.557825+00:00"},{"alias_kind":"pith_short_8","alias_value":"EPX5XKZF","created_at":"2026-07-05T11:12:35.557825+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07335","citing_title":"Defending Jailbreak Attacks on Large Language Models via Manifold Trajectory Kinetics","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00610","citing_title":"MemGraphRAG: Memory-based Multi-Agent System for Graph Retrieval-Augmented Generation","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19083","citing_title":"ProjLens: Unveiling the Role of Projectors in Multimodal Model Safety","ref_index":194,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06811","citing_title":"SkillTrojan: Backdoor Attacks on Skill-Based Agent Systems","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EPX5XKZFGRDZB4HP4OT7KDRPZG","json":"https://pith.science/pith/EPX5XKZFGRDZB4HP4OT7KDRPZG.json","graph_json":"https://pith.science/api/pith-number/EPX5XKZFGRDZB4HP4OT7KDRPZG/graph.json","events_json":"https://pith.science/api/pith-number/EPX5XKZFGRDZB4HP4OT7KDRPZG/events.json","paper":"https://pith.science/paper/EPX5XKZF"},"agent_actions":{"view_html":"https://pith.science/pith/EPX5XKZFGRDZB4HP4OT7KDRPZG","download_json":"https://pith.science/pith/EPX5XKZFGRDZB4HP4OT7KDRPZG.json","view_paper":"https://pith.science/paper/EPX5XKZF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.23793&json=true","fetch_graph":"https://pith.science/api/pith-number/EPX5XKZFGRDZB4HP4OT7KDRPZG/graph.json","fetch_events":"https://pith.science/api/pith-number/EPX5XKZFGRDZB4HP4OT7KDRPZG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EPX5XKZFGRDZB4HP4OT7KDRPZG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EPX5XKZFGRDZB4HP4OT7KDRPZG/action/storage_attestation","attest_author":"https://pith.science/pith/EPX5XKZFGRDZB4HP4OT7KDRPZG/action/author_attestation","sign_citation":"https://pith.science/pith/EPX5XKZFGRDZB4HP4OT7KDRPZG/action/citation_signature","submit_replication":"https://pith.science/pith/EPX5XKZFGRDZB4HP4OT7KDRPZG/action/replication_record"}},"created_at":"2026-07-05T11:12:35.557825+00:00","updated_at":"2026-07-05T11:12:35.557825+00:00"}