{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EOQWIQC5IUXM7OAI3CZJ42DRFR","short_pith_number":"pith:EOQWIQC5","schema_version":"1.0","canonical_sha256":"23a164405d452ecfb808d8b29e68712c60c5b82edab4ef73f14d53f210c323d0","source":{"kind":"arxiv","id":"2405.20947","version":5},"attestation_state":"computed","paper":{"title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Cho-Jui Hsieh, Ion Stoica, Justin Cui, Wei-Lin Chiang","submitted_at":"2024-05-31T15:44:33Z","abstract_excerpt":"Large Language Models (LLMs) require careful safety alignment to prevent malicious outputs. While significant research focuses on mitigating harmful content generation, the enhanced safety often come with the side effect of over-refusal, where LLMs may reject innocuous prompts and become less helpful. Although the issue of over-refusal has been empirically observed, a systematic measurement is challenging due to the difficulty of crafting prompts that can elicit the over-refusal behaviors of LLMs. This study proposes a novel method for automatically generating large-scale over-refusal datasets"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.20947","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-05-31T15:44:33Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"9bc65fbceadc09bcf1e03f65ad085dde583444dac09a35fe3bee812b8c64a518","abstract_canon_sha256":"d2794c0279639943d8439b5245008b00d9ed88fb37bb637131069b42c7adab6f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:21:33.340190Z","signature_b64":"ZUxRcnizGlJiwuCCFO8fntpisvWI5YlndVQhqtqjI/DOBb6JjYsQ0lK9GBvuzvUdQE3EWJIHpSWGOZBFZK38BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"23a164405d452ecfb808d8b29e68712c60c5b82edab4ef73f14d53f210c323d0","last_reissued_at":"2026-07-05T11:21:33.339537Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:21:33.339537Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Cho-Jui Hsieh, Ion Stoica, Justin Cui, Wei-Lin Chiang","submitted_at":"2024-05-31T15:44:33Z","abstract_excerpt":"Large Language Models (LLMs) require careful safety alignment to prevent malicious outputs. While significant research focuses on mitigating harmful content generation, the enhanced safety often come with the side effect of over-refusal, where LLMs may reject innocuous prompts and become less helpful. Although the issue of over-refusal has been empirically observed, a systematic measurement is challenging due to the difficulty of crafting prompts that can elicit the over-refusal behaviors of LLMs. This study proposes a novel method for automatically generating large-scale over-refusal datasets"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.20947","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.20947/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.20947","created_at":"2026-07-05T11:21:33.339619+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.20947v5","created_at":"2026-07-05T11:21:33.339619+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.20947","created_at":"2026-07-05T11:21:33.339619+00:00"},{"alias_kind":"pith_short_12","alias_value":"EOQWIQC5IUXM","created_at":"2026-07-05T11:21:33.339619+00:00"},{"alias_kind":"pith_short_16","alias_value":"EOQWIQC5IUXM7OAI","created_at":"2026-07-05T11:21:33.339619+00:00"},{"alias_kind":"pith_short_8","alias_value":"EOQWIQC5","created_at":"2026-07-05T11:21:33.339619+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":38,"internal_anchor_count":3,"sample":[{"citing_arxiv_id":"2607.07918","citing_title":"Efficient Safety Alignment of Language Models via Latent Personality Traits","ref_index":9,"is_internal_anchor":true},{"citing_arxiv_id":"2607.05842","citing_title":"Beyond Refusal: A Same-Lineage Study of Aligned and Abliterated LLMs for Vulnerability Analysis","ref_index":48,"is_internal_anchor":true},{"citing_arxiv_id":"2607.05355","citing_title":"Faithfulness to Refusal: A Causal Audit of Neuron Selectors","ref_index":8,"is_internal_anchor":true},{"citing_arxiv_id":"2606.09843","citing_title":"An LLM-Native Psychometric Instrument Reveals a Self-Report--Behavior Gap Across 25 Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26396","citing_title":"At the Edge of Understanding: Sparse Autoencoders Trace The Limits of Transformer Generalization","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02047","citing_title":"OpenSafeIntent: Evaluating Intent-Calibrated Safe Completion Across Dual-Use Prompt Sets","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11316","citing_title":"Sch\\\"utzen: Evaluating LLM Safety in Bulgarian and German Contexts","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09697","citing_title":"PsychoSafe: Eliciting Psychologically-Informed Refusals in Large Language Models","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08381","citing_title":"Auditing Proprietary Alignment in Large Language Models: A Comparative Framework Without a Ground-Truth Standard","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05523","citing_title":"CHASE: Adversarial Red-Blue Teaming for Improving LLM Safety using Reinforcement Learning","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03330","citing_title":"FLIPS: Instance-Fingerprinting for LLMs via Pseudo-random Sequences","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02965","citing_title":"Designing for Doubt: The Case for Informed Abstention in Autonomous Agents","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00600","citing_title":"Understanding the Self-Reflection Mechanisms of LLMs through Biased Attitude Associations","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31748","citing_title":"Addressing Over-Refusal in LLMs with Competing Rewards","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12429","citing_title":"Muse Spark Safety & Preparedness Report","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24154","citing_title":"Palette: A Modular, Controllable, and Efficient Framework for On-demand Authorized Safety Alignment Relaxation in LLMs","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24552","citing_title":"Ellipsoid Control: A White-list Jailbreak Defense via Benign Latent Modeling","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28647","citing_title":"The Ethics of LLM Sandbox and Persona Dynamics","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29659","citing_title":"Opir: Efficient Multi-Task Safety Classification for Toxicity, Jailbreaks, Hate Speech, and Harmful Content","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30693","citing_title":"Triaging Threats to Specialized Guardrails","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00686","citing_title":"Dialectics of Alignment: Harnessing Unsafe Knowledge for Dynamic Safety Routing","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21296","citing_title":"Discriminatory Compliance: How LLMs Answer Queries from Protected Groups","ref_index":104,"is_internal_anchor":false},{"citing_arxiv_id":"2503.02574","citing_title":"LLM-Safety Evaluations Lack Robustness","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21545","citing_title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16282","citing_title":"Taxonomy and Consistency Analysis of Safety Benchmarks for AI Agents","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EOQWIQC5IUXM7OAI3CZJ42DRFR","json":"https://pith.science/pith/EOQWIQC5IUXM7OAI3CZJ42DRFR.json","graph_json":"https://pith.science/api/pith-number/EOQWIQC5IUXM7OAI3CZJ42DRFR/graph.json","events_json":"https://pith.science/api/pith-number/EOQWIQC5IUXM7OAI3CZJ42DRFR/events.json","paper":"https://pith.science/paper/EOQWIQC5"},"agent_actions":{"view_html":"https://pith.science/pith/EOQWIQC5IUXM7OAI3CZJ42DRFR","download_json":"https://pith.science/pith/EOQWIQC5IUXM7OAI3CZJ42DRFR.json","view_paper":"https://pith.science/paper/EOQWIQC5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.20947&json=true","fetch_graph":"https://pith.science/api/pith-number/EOQWIQC5IUXM7OAI3CZJ42DRFR/graph.json","fetch_events":"https://pith.science/api/pith-number/EOQWIQC5IUXM7OAI3CZJ42DRFR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EOQWIQC5IUXM7OAI3CZJ42DRFR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EOQWIQC5IUXM7OAI3CZJ42DRFR/action/storage_attestation","attest_author":"https://pith.science/pith/EOQWIQC5IUXM7OAI3CZJ42DRFR/action/author_attestation","sign_citation":"https://pith.science/pith/EOQWIQC5IUXM7OAI3CZJ42DRFR/action/citation_signature","submit_replication":"https://pith.science/pith/EOQWIQC5IUXM7OAI3CZJ42DRFR/action/replication_record"}},"created_at":"2026-07-05T11:21:33.339619+00:00","updated_at":"2026-07-05T11:21:33.339619+00:00"}