{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:LILDZUIXAFSBHAWZZH4TTIISHY","short_pith_number":"pith:LILDZUIX","schema_version":"1.0","canonical_sha256":"5a163cd11701641382d9c9f939a1123e2252cad7e7ce315a7c60f381232d6b2a","source":{"kind":"arxiv","id":"2310.17389","version":1},"attestation_state":"computed","paper":{"title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jingbo Shang, Yangkun Wang, Yongqi Tong, Yujia Wang, Yuxin Guo, Zihan Wang, Zi Lin","submitted_at":"2023-10-26T13:35:41Z","abstract_excerpt":"Despite remarkable advances that large language models have achieved in chatbots, maintaining a non-toxic user-AI interactive environment has become increasingly critical nowadays. However, previous efforts in toxicity detection have been mostly based on benchmarks derived from social media content, leaving the unique challenges inherent to real-world user-AI interactions insufficiently explored. In this work, we introduce ToxicChat, a novel benchmark based on real user queries from an open-source chatbot. This benchmark contains the rich, nuanced phenomena that can be tricky for current toxic"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.17389","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-26T13:35:41Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"194a354805f8dd73e12b6e0a043b916f7ae721b3b396ce3be2fab8453a94a5e0","abstract_canon_sha256":"aa610e537af1a3c3323130eaa696546131f4ecc92d076397a51d9263b273ac10"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:05:32.936497Z","signature_b64":"lOIdAKf0yjYeVWFQW9idT+hnCmoFyPTJ6/euHpn5oWNW9VY++Qsx4sJSGvD4NeiKwbNRW6Qsk6N0loHfEJLECw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5a163cd11701641382d9c9f939a1123e2252cad7e7ce315a7c60f381232d6b2a","last_reissued_at":"2026-07-05T07:05:32.935974Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:05:32.935974Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jingbo Shang, Yangkun Wang, Yongqi Tong, Yujia Wang, Yuxin Guo, Zihan Wang, Zi Lin","submitted_at":"2023-10-26T13:35:41Z","abstract_excerpt":"Despite remarkable advances that large language models have achieved in chatbots, maintaining a non-toxic user-AI interactive environment has become increasingly critical nowadays. However, previous efforts in toxicity detection have been mostly based on benchmarks derived from social media content, leaving the unique challenges inherent to real-world user-AI interactions insufficiently explored. In this work, we introduce ToxicChat, a novel benchmark based on real user queries from an open-source chatbot. This benchmark contains the rich, nuanced phenomena that can be tricky for current toxic"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.17389","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.17389/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.17389","created_at":"2026-07-05T07:05:32.936025+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.17389v1","created_at":"2026-07-05T07:05:32.936025+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.17389","created_at":"2026-07-05T07:05:32.936025+00:00"},{"alias_kind":"pith_short_12","alias_value":"LILDZUIXAFSB","created_at":"2026-07-05T07:05:32.936025+00:00"},{"alias_kind":"pith_short_16","alias_value":"LILDZUIXAFSBHAWZ","created_at":"2026-07-05T07:05:32.936025+00:00"},{"alias_kind":"pith_short_8","alias_value":"LILDZUIX","created_at":"2026-07-05T07:05:32.936025+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.02079","citing_title":"HaloGuard 1.0: An Open Weights Constitutional Classifier for Multilingual AI Safety","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07335","citing_title":"Defending Jailbreak Attacks on Large Language Models via Manifold Trajectory Kinetics","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23598","citing_title":"When Youth Enter the Algorithmic Wild: Discovering and Understanding Potentially Harmful Teen Videos on Douyin and Kwai","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2510.13727","citing_title":"From Refusal to Recovery: A Control-Theoretic Approach to Generative AI Guardrails","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2407.21772","citing_title":"ShieldGemma: Generative AI Content Moderation Based on Gemma","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2406.18495","citing_title":"WildGuard: Open One-Stop Moderation Tools for Safety Risks, Jailbreaks, and Refusals of LLMs","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10639","citing_title":"Navigating the Sea of LLM Evaluation: Investigating Bias in Toxicity Benchmarks","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22154","citing_title":"Reliable Self-Harm Risk Screening via Adaptive Multi-Agent LLM Systems","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20945","citing_title":"Breaking Bad: Interpretability-Based Safety Audits of State-of-the-Art LLMs","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06833","citing_title":"FedDetox: Robust Federated SLM Alignment via On-Device Data Sanitization","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07655","citing_title":"Guardian-as-an-Advisor: Advancing Next-Generation Guardian Models for Trustworthy LLMs","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18519","citing_title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16542","citing_title":"TWGuard: A Case Study of LLM Safety Guardrails for Localized Linguistic Contexts","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LILDZUIXAFSBHAWZZH4TTIISHY","json":"https://pith.science/pith/LILDZUIXAFSBHAWZZH4TTIISHY.json","graph_json":"https://pith.science/api/pith-number/LILDZUIXAFSBHAWZZH4TTIISHY/graph.json","events_json":"https://pith.science/api/pith-number/LILDZUIXAFSBHAWZZH4TTIISHY/events.json","paper":"https://pith.science/paper/LILDZUIX"},"agent_actions":{"view_html":"https://pith.science/pith/LILDZUIXAFSBHAWZZH4TTIISHY","download_json":"https://pith.science/pith/LILDZUIXAFSBHAWZZH4TTIISHY.json","view_paper":"https://pith.science/paper/LILDZUIX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.17389&json=true","fetch_graph":"https://pith.science/api/pith-number/LILDZUIXAFSBHAWZZH4TTIISHY/graph.json","fetch_events":"https://pith.science/api/pith-number/LILDZUIXAFSBHAWZZH4TTIISHY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LILDZUIXAFSBHAWZZH4TTIISHY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LILDZUIXAFSBHAWZZH4TTIISHY/action/storage_attestation","attest_author":"https://pith.science/pith/LILDZUIXAFSBHAWZZH4TTIISHY/action/author_attestation","sign_citation":"https://pith.science/pith/LILDZUIXAFSBHAWZZH4TTIISHY/action/citation_signature","submit_replication":"https://pith.science/pith/LILDZUIXAFSBHAWZZH4TTIISHY/action/replication_record"}},"created_at":"2026-07-05T07:05:32.936025+00:00","updated_at":"2026-07-05T07:05:32.936025+00:00"}