{"as_of":"2026-08-10T07:30:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:582c4c7f05ae0ed854a0c236c690defeea8e18386a53868d36b02087fce90920","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":24,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":24,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":24,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":24,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T10:23:06.168953Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T14:48:32.654187Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":"2310.17389","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-07-03T14:48:32.654187Z","title":"Toxic- chat: Unveiling hidden challenges of toxicity detec- tion in real-world user-ai conversation","venue":null,"work_id":"934340c1-de04-43a3-af48-deb728f15d44","year":2023},"citing_paper":{"arxiv_id":"2406.18495","last_updated":"2024-12-09T20:21:56Z","snapshot_observed_at":"2026-08-06T04:32:59.039501Z","submitted_at":"2024-06-26T16:58:20Z","title":"WildGuard: Open One-Stop Moderation Tools for Safety Risks, Jailbreaks, and Refusals of LLMs","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-17T16:25:14.744887Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2406.18495"},"observation_digest":"sha256:483ad10a1fe8577c9dc9cc91ddb5a3b1fd04c38d15a05f0185f8be08e73bb36d","observation_id":"4ae480b6-9c61-470c-a90d-6399246e790d","resolution":{"observed_at":"2026-05-17T16:25:14.866587Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":"2310.17389","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-07-03T14:48:32.654187Z","title":"Toxic- chat: Unveiling hidden challenges of toxicity detec- tion in real-world user-ai conversation","venue":null,"work_id":"934340c1-de04-43a3-af48-deb728f15d44","year":2023},"citing_paper":{"arxiv_id":"2407.21772","last_updated":"2024-08-04T22:13:39Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-31T17:48:14Z","title":"ShieldGemma: Generative AI Content Moderation Based on Gemma","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-20T13:17:39.444002Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2407.21772"},"observation_digest":"sha256:c73b82480992cc9585821e08123b165b57204df5d269c2186303a3866a38c17d","observation_id":"a6049c2b-aeaa-44a8-8685-d434ac56dbb4","resolution":{"observed_at":"2026-05-20T13:17:39.523130Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-08-08T10:23:06.168953Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.08142","last_updated":"2025-02-12T05:48:57Z","snapshot_observed_at":"2026-08-09T00:12:26.816440Z","submitted_at":"2025-02-12T05:48:57Z","title":"Bridging the Safety Gap: A Guardrail Pipeline for Trustworthy LLM Inferences","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-08T10:23:06.168953Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2502.08142"},"observation_digest":"sha256:9624a9c56dfc075f31a0b3e1be2cad380a179f60e693bf412307842762145765","observation_id":"57b6ae12-f0d5-406f-b47f-1ed7be08b990","resolution":{"observed_at":"2026-08-08T10:23:06.168953Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-08-07T14:52:28.909375Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:28.909375Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:b3b557e4a3ce3210b7dbdf559cd9865f84f6cd0094113f6369f14b53e0c0e9ce","observation_id":"e773cb32-b65a-4db0-8695-1cd4132b6e08","resolution":{"observed_at":"2026-08-07T14:52:28.909375Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-08-07T13:16:15.795060Z","title":"Thomas Hartvigsen, Saadia Gabriel, Hamid Palangi, Maarten Sap, Dipankar Ray, and Ece Kamar","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.23836","last_updated":"2025-07-16T11:25:40Z","snapshot_observed_at":"2026-08-07T13:08:50.821464Z","submitted_at":"2025-05-28T12:03:09Z","title":"Large Language Models Often Know When They Are Being Evaluated","version":3},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-07T13:16:15.795060Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2505.23836"},"observation_digest":"sha256:4bdebc64aaf034cc6774ae5fbb7946573b84e6bfbd1d898cb773f17c16f5f2d5","observation_id":"96d8ff20-356b-4be2-9aba-bb9b38955dea","resolution":{"observed_at":"2026-08-07T13:16:15.795060Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-08-07T10:28:54.802416Z","title":"Lin et al., ‘ToxicChat: Unveiling Hid- den Challenges of Toxicity Detection in Real-World User-AI Conversation’, Oct","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06391","last_updated":"2025-06-05T16:53:29Z","snapshot_observed_at":"2026-08-08T01:20:47.850082Z","submitted_at":"2025-06-05T16:53:29Z","title":"From Rogue to Safe AI: The Role of Explicit Refusals in Aligning LLMs with International Humanitarian Law","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:54.802416Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2506.06391"},"observation_digest":"sha256:48e3342fbb74225da5d9b68a51723713e66a9e7bdfaebdda17d90d3b6528ea7c","observation_id":"c0f879a3-92f5-496f-abec-9b8f05638b4d","resolution":{"observed_at":"2026-08-07T10:28:54.802416Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-08-07T10:17:26.647489Z","title":"Toxicchat: Unveiling hidden challenges of toxicity detection in real- world user-ai conversation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.11094","last_updated":"2025-10-30T06:22:33Z","snapshot_observed_at":"2026-08-07T10:11:06.747781Z","submitted_at":"2025-06-06T05:50:50Z","title":"The Scales of Justitia: A Comprehensive Survey on Safety Evaluation of LLMs","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T10:17:26.647489Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2506.11094"},"observation_digest":"sha256:3b00593451c8669d2cbe2413b31f96c2f8515d2b9813791e43795e6cd3897c8f","observation_id":"899debdd-8dbc-4652-b50f-8fe1d3245bd9","resolution":{"observed_at":"2026-08-07T10:17:26.647489Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-08-06T21:45:02.699513Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation , 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23706","last_updated":"2025-06-30T10:29:42Z","snapshot_observed_at":"2026-08-08T23:28:19.101327Z","submitted_at":"2025-06-30T10:29:42Z","title":"Attestable Audits: Verifiable AI Safety Benchmarks Using Trusted Execution Environments","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-06T21:45:02.699513Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2506.23706"},"observation_digest":"sha256:56fbc8bb30f0ba83c19c95c507c52b68b3b1a768981bddbea43ff3c04c14f642","observation_id":"a59b22d2-ab12-433d-8fca-fcf93f1968f3","resolution":{"observed_at":"2026-08-06T21:45:02.699513Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-08-06T17:03:15.999906Z","title":"Toxicchat: Unveiling hidden challenges of toxicity detection in real-world user-ai conversation","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.11878","last_updated":"2026-07-06T01:46:44Z","snapshot_observed_at":"2026-08-06T16:57:07.977935Z","submitted_at":"2025-07-16T03:48:03Z","title":"LLMs Encode Harmfulness and Refusal Separately","version":5},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T17:03:15.999906Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2507.11878"},"observation_digest":"sha256:247796c6e56ab771d68acb786e5cf86718a523b3eeb57276cb607f3fcc5ba81d","observation_id":"8945f4db-2a41-4c61-9a11-ef0ac1d66d3e","resolution":{"observed_at":"2026-08-06T17:03:15.999906Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":"2310.17389","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-07-03T14:48:32.654187Z","title":"Toxic- chat: Unveiling hidden challenges of toxicity detec- tion in real-world user-ai conversation","venue":null,"work_id":"934340c1-de04-43a3-af48-deb728f15d44","year":2023},"citing_paper":{"arxiv_id":"2510.13727","last_updated":"2026-05-19T04:39:26Z","snapshot_observed_at":"2026-08-02T19:37:00.796741Z","submitted_at":"2025-10-15T16:30:57Z","title":"From Refusal to Recovery: A Control-Theoretic Approach to Generative AI Guardrails","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-21T20:42:40.823721Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2510.13727"},"observation_digest":"sha256:b8ad055511d331c65bc210bf24ac4377d7bdbc9f0384a723524b28a6f41c301f","observation_id":"7d08457f-5a13-4da3-af0c-6a428b8406fa","resolution":{"observed_at":"2026-05-21T20:44:22.044971Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-08-03T05:07:09.235559Z","title":"Haotian Liu, Chunyuan Li, Qingyang Wu, and Yong Jae Lee","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.03328","last_updated":"2026-05-27T13:01:56Z","snapshot_observed_at":"2026-08-04T00:07:19.801811Z","submitted_at":"2026-02-03T09:56:20Z","title":"GuardReasoner-Omni: A Reasoning-based Multi-modal Guardrail for Text, Image, Video, and Audio","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-03T05:07:09.235559Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2602.03328"},"observation_digest":"sha256:ac90c3d8529d83de6d4b2391fc5c8dea4fccedd85acd4525aa0ddc2925befde2","observation_id":"fe7c1cb3-98e9-4202-af1b-118f250474bf","resolution":{"observed_at":"2026-08-03T05:07:09.235559Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-08-03T01:17:12.009201Z","title":"Toxicchat: Unveiling hidden challenges of toxicity detection in real-world user-ai conver- sation.arXiv preprint arXiv:2310.17389, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.10388","last_updated":"2026-05-29T05:05:29Z","snapshot_observed_at":"2026-08-07T17:40:15.715656Z","submitted_at":"2026-02-11T00:23:13Z","title":"Less is Enough: Synthesizing Diverse Data in LLM Feature Space with Sparse Autoencoders","version":4},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-03T01:17:12.009201Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2602.10388"},"observation_digest":"sha256:e191c895c155f6d21af736182044b30f7d4a2b037dcb10ee197f1b5bcdf9678d","observation_id":"86848d71-d41c-4229-aecd-7ea9e6c58cbc","resolution":{"observed_at":"2026-08-03T01:17:12.009201Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":"2310.17389","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-07-03T14:48:32.654187Z","title":"Toxic- chat: Unveiling hidden challenges of toxicity detec- tion in real-world user-ai conversation","venue":null,"work_id":"934340c1-de04-43a3-af48-deb728f15d44","year":2023},"citing_paper":{"arxiv_id":"2604.06833","last_updated":"2026-04-08T08:51:46Z","snapshot_observed_at":"2026-07-06T22:55:15.791334Z","submitted_at":"2026-04-08T08:51:46Z","title":"FedDetox: Robust Federated SLM Alignment via On-Device Data Sanitization","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-10T17:22:40.613937Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2604.06833"},"observation_digest":"sha256:2f61973fc6061b67f0f297b309d36e97ca0c21f27816472ebd7f9a5824d28c2d","observation_id":"f7e3cf6a-63e5-4fe4-b5bc-3396ff7d194c","resolution":{"observed_at":"2026-05-11T06:56:01.782343Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":"2310.17389","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-07-03T14:48:32.654187Z","title":"Toxic- chat: Unveiling hidden challenges of toxicity detec- tion in real-world user-ai conversation","venue":null,"work_id":"934340c1-de04-43a3-af48-deb728f15d44","year":2023},"citing_paper":{"arxiv_id":"2604.07655","last_updated":"2026-04-08T23:47:29Z","snapshot_observed_at":"2026-08-02T09:41:13.361711Z","submitted_at":"2026-04-08T23:47:29Z","title":"Guardian-as-an-Advisor: Advancing Next-Generation Guardian Models for Trustworthy LLMs","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-10T17:27:13.339411Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2604.07655"},"observation_digest":"sha256:e56d4b6be7fd791b401e21e5bc96b175603a50cbb7b2168dc4d579a556305f86","observation_id":"14c65156-fe1a-4726-8f6c-d6fff637a2ff","resolution":{"observed_at":"2026-05-11T06:46:37.658425Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":"2310.17389","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-07-03T14:48:32.654187Z","title":"Toxic- chat: Unveiling hidden challenges of toxicity detec- tion in real-world user-ai conversation","venue":null,"work_id":"934340c1-de04-43a3-af48-deb728f15d44","year":2023},"citing_paper":{"arxiv_id":"2604.16542","last_updated":"2026-04-17T01:55:37Z","snapshot_observed_at":"2026-08-01T22:59:54.227625Z","submitted_at":"2026-04-17T01:55:37Z","title":"TWGuard: A Case Study of LLM Safety Guardrails for Localized Linguistic Contexts","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-10T09:07:57.713675Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2604.16542"},"observation_digest":"sha256:76dbd8241ce1cf8b2ddd4f5dee33eec1d28d44d73297d66ef9d3b4918792a4ba","observation_id":"f3e2d616-a3f3-4103-a614-aa060b0a2e05","resolution":{"observed_at":"2026-05-10T09:08:25.512197Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":"2310.17389","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-07-03T14:48:32.654187Z","title":"Toxic- chat: Unveiling hidden challenges of toxicity detec- tion in real-world user-ai conversation","venue":null,"work_id":"934340c1-de04-43a3-af48-deb728f15d44","year":2023},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:e93334c7b8bc313045944d2d43c4f955d2c1f4b14bf1a21b14a21910b1dd9919","observation_id":"d5b2a56b-da7c-4d7e-99ac-83fb41c9db39","resolution":{"observed_at":"2026-05-10T12:20:23.136237Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":"2310.17389","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-07-03T14:48:32.654187Z","title":"Toxic- chat: Unveiling hidden challenges of toxicity detec- tion in real-world user-ai conversation","venue":null,"work_id":"934340c1-de04-43a3-af48-deb728f15d44","year":2023},"citing_paper":{"arxiv_id":"2604.20945","last_updated":"2026-04-22T16:51:49Z","snapshot_observed_at":"2026-07-06T23:07:36.996629Z","submitted_at":"2026-04-22T16:51:49Z","title":"Breaking Bad: Interpretability-Based Safety Audits of State-of-the-Art LLMs","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T00:32:00.789721Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2604.20945"},"observation_digest":"sha256:2e96819d34da0c79577d1d8ac87cc595b889f5e303a2680b7ed8a1377203d04c","observation_id":"e8af5d2c-560f-4e6f-aa37-e9d5cb70633c","resolution":{"observed_at":"2026-05-11T13:46:05.501341Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":"2310.17389","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-07-03T14:48:32.654187Z","title":"Toxic- chat: Unveiling hidden challenges of toxicity detec- tion in real-world user-ai conversation","venue":null,"work_id":"934340c1-de04-43a3-af48-deb728f15d44","year":2023},"citing_paper":{"arxiv_id":"2604.22154","last_updated":"2026-04-24T01:52:54Z","snapshot_observed_at":"2026-07-06T23:08:37.811548Z","submitted_at":"2026-04-24T01:52:54Z","title":"Reliable Self-Harm Risk Screening via Adaptive Multi-Agent LLM Systems","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-08T12:41:53.950049Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2604.22154"},"observation_digest":"sha256:7bd1fec7b2110ac7ce5a03dc04323019a10f384828a882b21d7065262e512fd5","observation_id":"2d758f06-8adc-4c14-8c4d-2f363f1f20ad","resolution":{"observed_at":"2026-05-11T19:06:09.741391Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":"2310.17389","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-07-03T14:48:32.654187Z","title":"Toxic- chat: Unveiling hidden challenges of toxicity detec- tion in real-world user-ai conversation","venue":null,"work_id":"934340c1-de04-43a3-af48-deb728f15d44","year":2023},"citing_paper":{"arxiv_id":"2605.10639","last_updated":"2026-05-11T14:27:39Z","snapshot_observed_at":"2026-08-02T13:54:55.570407Z","submitted_at":"2026-05-11T14:27:39Z","title":"Navigating the Sea of LLM Evaluation: Investigating Bias in Toxicity Benchmarks","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-12T05:28:45.453455Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2605.10639"},"observation_digest":"sha256:7b37bf89efeedceaa702bbce41c7f83cb4cdf6746755e224c38a980a104d4668","observation_id":"01f579b9-361c-4712-a23f-9db28b5a9661","resolution":{"observed_at":"2026-05-12T05:31:23.946872Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":"2310.17389","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-07-03T14:48:32.654187Z","title":"Toxic- chat: Unveiling hidden challenges of toxicity detec- tion in real-world user-ai conversation","venue":null,"work_id":"934340c1-de04-43a3-af48-deb728f15d44","year":2023},"citing_paper":{"arxiv_id":"2605.23598","last_updated":"2026-05-22T13:06:46Z","snapshot_observed_at":"2026-08-05T16:03:21.062446Z","submitted_at":"2026-05-22T13:06:46Z","title":"When Youth Enter the Algorithmic Wild: Discovering and Understanding Potentially Harmful Teen Videos on Douyin and Kwai","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-25T04:20:27.456558Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2605.23598"},"observation_digest":"sha256:d15ada0d47b59733117423cb495dddb9dfa8b0463b9ea8ab57b821d34ed3b0e7","observation_id":"700c0b50-026d-4736-b3d9-31e4fa83de6f","resolution":{"observed_at":"2026-05-25T04:25:19.788374Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":"2310.17389","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-07-03T14:48:32.654187Z","title":"Toxic- chat: Unveiling hidden challenges of toxicity detec- tion in real-world user-ai conversation","venue":null,"work_id":"934340c1-de04-43a3-af48-deb728f15d44","year":2023},"citing_paper":{"arxiv_id":"2606.07335","last_updated":"2026-06-05T14:49:26Z","snapshot_observed_at":"2026-07-06T23:47:00.425715Z","submitted_at":"2026-06-05T14:49:26Z","title":"Defending Jailbreak Attacks on Large Language Models via Manifold Trajectory Kinetics","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-06-27T21:55:48.561400Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2606.07335"},"observation_digest":"sha256:5552f63ec82d94e7e11dbfbf4e2e8b267429174c114aad0c23601557fd16d938","observation_id":"c61f411c-1406-4b77-8914-6f10225690b0","resolution":{"observed_at":"2026-07-02T17:37:14.897253Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":"2310.17389","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-07-03T14:48:32.654187Z","title":"Toxic- chat: Unveiling hidden challenges of toxicity detec- tion in real-world user-ai conversation","venue":null,"work_id":"934340c1-de04-43a3-af48-deb728f15d44","year":2023},"citing_paper":{"arxiv_id":"2607.02079","last_updated":"2026-07-02T12:21:16Z","snapshot_observed_at":"2026-08-06T21:17:29.886069Z","submitted_at":"2026-07-02T12:21:16Z","title":"HaloGuard 1.0: An Open Weights Constitutional Classifier for Multilingual AI Safety","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-03T14:38:55.045628Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2607.02079"},"observation_digest":"sha256:b08b1ba46beef5c8eaa0ada7b9907b8eb8da0c61f2652d711e9097303c11e417","observation_id":"59558a22-481c-44f1-a762-506f6272a304","resolution":{"observed_at":"2026-07-03T14:48:32.655714Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-08-02T14:52:12.591220Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.22545","last_updated":"2026-05-06T10:14:30Z","snapshot_observed_at":"2026-08-09T18:04:23.587872Z","submitted_at":"2026-05-06T10:14:30Z","title":"Semalith v1.4: A Calibrated 184M Safety Classifier Achieving State-of-the-Art Prompt-Injection Detection at 44x Fewer Parameters than Llama-Guard-3-8B","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-02T14:52:12.591220Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2607.22545"},"observation_digest":"sha256:9c182b7d34692aa1c05d6dfddc0cf3c00d8af9c388811cfa1eb26955c71ad7e4","observation_id":"048e4edc-5e2d-46cd-aa71-38c0bc50b6ab","resolution":{"observed_at":"2026-08-02T14:52:12.591220Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-08-04T01:07:16.653669Z","title":"ToxicChat: Unveiling hidden challenges of toxicity detection in real-world user-AI conversation.arXiv preprint arXiv:2310.17389,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00180","last_updated":"2026-08-04T02:59:21Z","snapshot_observed_at":"2026-08-07T23:09:41.591880Z","submitted_at":"2026-07-31T18:05:25Z","title":"A Constitution-Grid Instrument for Data-Efficient RL Alignment (C-Guard)","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-04T01:07:16.653669Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2608.00180"},"observation_digest":"sha256:351a3ca9a1eadb6060d38da5bd48e828bf44fcc7ef1f71cecba5c54e4de799f1","observation_id":"2fd0eeb0-ba99-451f-a1ed-ebdbb1a87b24","resolution":{"observed_at":"2026-08-04T01:07:16.653669Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2310.17389/citation-record","integrity":"/paper/2310.17389/integrity","json":"/paper/2310.17389/citation-record.json","paper":"/paper/2310.17389"},"outbound":[],"paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 24 inbound Pith citation observations for arXiv:2310.17389."}