{"as_of":"2026-08-09T06:33:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:7374c2eb477ccdfcb9f8e8bcb8a0eaf3670ac5fb1b53c4dd76321587ef399f41","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":35,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":35,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":35,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":35,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:52:42.777127Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":19,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2309.07045","doi":"10.48550/arxiv.2309.07045","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SafetyBench: Evaluating the Safety of Large Language Models with Multiple Choice Questions","venue":"arXiv (Cornell University)","work_id":"e26de3a1-77ec-4538-b0f3-09cea1fc3e90","year":2023},"citing_paper":{"arxiv_id":"2406.12793","last_updated":"2024-07-30T03:58:11Z","snapshot_observed_at":"2026-08-07T13:56:34.167869Z","submitted_at":"2024-06-18T16:58:21Z","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-11T08:08:09.444352Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2406.12793"},"observation_digest":"sha256:2d55a8cb113f0663937caae4d49bdfdf9c5d26764fe4ff499b4d03d71096a910","observation_id":"9af1f249-4f1a-4f1d-b322-b462cff38150","resolution":{"observed_at":"2026-05-11T08:08:09.724948Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2309.07045","doi":"10.48550/arxiv.2309.07045","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SafetyBench: Evaluating the Safety of Large Language Models with Multiple Choice Questions","venue":"arXiv (Cornell University)","work_id":"e26de3a1-77ec-4538-b0f3-09cea1fc3e90","year":2023},"citing_paper":{"arxiv_id":"2407.04295","last_updated":"2024-08-30T11:57:47Z","snapshot_observed_at":"2026-08-04T23:34:13.332065Z","submitted_at":"2024-07-05T06:57:30Z","title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","version":2},"reference_index":115,"source":"pdf_text","source_observed_at":"2026-05-15T02:20:44.368219Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2407.04295"},"observation_digest":"sha256:b36c69c840a61d3151345f3cf9b82beb81a10e46360034f7370835a59b5b1e8c","observation_id":"fffc98aa-9840-491a-89b7-8920516259bb","resolution":{"observed_at":"2026-05-15T02:20:44.493758Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-07T14:52:31.148227Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:31.148227Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:347d0a068073efe5bbbab1911e7f400fac910913f0c9a5a24634d0d109dc3412","observation_id":"a444735e-47da-4991-9cee-16ece478ec49","resolution":{"observed_at":"2026-08-07T14:52:31.148227Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-07T14:52:42.777127Z","title":"Safetybench: Evaluating the safety of large language models with multiple choice questions.arXiv preprint arXiv:2309.07045, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17407","last_updated":"2025-05-23T02:46:18Z","snapshot_observed_at":"2026-08-08T01:26:05.514174Z","submitted_at":"2025-05-23T02:46:18Z","title":"Language Matters: How Do Multilingual Input and Reasoning Paths Affect Large Reasoning Models?","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T14:52:42.777127Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2505.17407"},"observation_digest":"sha256:a97b09ae66ecf8041123a48b18bba22f913a7462a094f9abdaa12b693c536d0a","observation_id":"61a96958-5215-45cd-a6be-9a719353b435","resolution":{"observed_at":"2026-08-07T14:52:42.777127Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-07T13:00:59.091799Z","title":"Safetybench: Evaluating the safety of large language models.arXiv preprint arXiv:2309.07045, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.22959","last_updated":"2025-05-29T01:02:53Z","snapshot_observed_at":"2026-08-07T12:54:11.901690Z","submitted_at":"2025-05-29T01:02:53Z","title":"LLM-based HSE Compliance Assessment: Benchmark, Performance, and Advancements","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T13:00:59.091799Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2505.22959"},"observation_digest":"sha256:5a4074e1a298413e821ee4928ba6f148108f5f5891588283da86e7d85bc2d2f6","observation_id":"fad597c3-b1dd-45f1-b0ff-93fac7ab0d79","resolution":{"observed_at":"2026-08-07T13:00:59.091799Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-07T14:15:16.805123Z","title":"SafetyBench: Evaluating the safety of large language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23793","last_updated":"2025-05-26T08:39:14Z","snapshot_observed_at":"2026-08-08T01:18:19.713600Z","submitted_at":"2025-05-26T08:39:14Z","title":"USB: A Comprehensive and Unified Safety Evaluation Benchmark for Multimodal Large Language Models","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:16.805123Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2505.23793"},"observation_digest":"sha256:526332e1c4a45d4ac78325819b9ff6ba6ce5c3caa214d495d33b86cdec97f2e7","observation_id":"fe0f0a71-571a-48fb-be9b-dc635c57f7cb","resolution":{"observed_at":"2026-08-07T14:15:16.805123Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-07T13:16:15.735454Z","title":"Bertie Vidgen, Nino Scherrer, Hannah Rose Kirk, Rebecca Qian, Anand Kannappan, Scott A","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23836","last_updated":"2025-07-16T11:25:40Z","snapshot_observed_at":"2026-08-07T13:08:50.821464Z","submitted_at":"2025-05-28T12:03:09Z","title":"Large Language Models Often Know When They Are Being Evaluated","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T13:16:15.735454Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2505.23836"},"observation_digest":"sha256:f7609645c64576e571bd0ac3edaca656bbba0a9912bd76c8c2f47b64c9efc3d4","observation_id":"8d2f4893-47b3-4137-af3c-f0aa5f4dec6e","resolution":{"observed_at":"2026-08-07T13:16:15.735454Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-07T11:50:54.874110Z","title":"Safetybench: Evaluating the safety of large language models.arXiv preprint arXiv:2309.07045, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01252","last_updated":"2025-06-02T02:01:40Z","snapshot_observed_at":"2026-08-09T04:45:00.469971Z","submitted_at":"2025-06-02T02:01:40Z","title":"MTCMB: A Multi-Task Benchmark Framework for Evaluating LLMs on Knowledge, Reasoning, and Safety in Traditional Chinese Medicine","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T11:50:54.874110Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2506.01252"},"observation_digest":"sha256:ea8fe85c9869e061dda9d196b07b93dbd2121c661bda43b74e1678329c9c4899","observation_id":"e8ab284a-3778-40f8-9bf3-8bd0b1e2834d","resolution":{"observed_at":"2026-08-07T11:50:54.874110Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-07T05:19:02.714368Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.08399","last_updated":"2025-06-11T06:57:37Z","snapshot_observed_at":"2026-08-08T08:31:17.317755Z","submitted_at":"2025-06-10T03:13:50Z","title":"SafeCoT: Improving VLM Safety with Minimal Reasoning","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-07T05:19:02.714368Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2506.08399"},"observation_digest":"sha256:7a278dca078847efa776364f5b71b1594b7e8e5a36fc55f5bb9f642e1faab019","observation_id":"a58398f4-79b8-4b32-8d86-a01a50d3b514","resolution":{"observed_at":"2026-08-07T05:19:02.714368Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-06T23:49:54.958475Z","title":"arXiv preprint arXiv:2309.07045","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.16322","last_updated":"2025-06-19T13:56:41Z","snapshot_observed_at":"2026-08-07T20:46:18.050527Z","submitted_at":"2025-06-19T13:56:41Z","title":"PL-Guard: Benchmarking Language Model Safety for Polish","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T23:49:54.958475Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2506.16322"},"observation_digest":"sha256:b66538eb30ac6a23f741de611608904cc947147ee0a21456ad0295614c25a573","observation_id":"5152cbc0-bd5b-4048-866a-a5f5ef1b9b59","resolution":{"observed_at":"2026-08-06T23:49:54.958475Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-06T10:31:06.665914Z","title":"negative_impacts","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.23718","last_updated":"2025-07-31T16:52:21Z","snapshot_observed_at":"2026-08-09T03:31:36.620563Z","submitted_at":"2025-07-31T16:52:21Z","title":"Informing AI Risk Assessment with News Media: Analyzing National and Political Variation in the Coverage of AI Risks","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T10:31:06.665914Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2507.23718"},"observation_digest":"sha256:cf1a9ddcbaa9c2a6dd4b43dcdf595fdc0f9b70036b0d230da1521b7c3e92f95b","observation_id":"ff373dd7-bf4a-417b-ab03-8b441f3d62af","resolution":{"observed_at":"2026-08-06T10:31:06.665914Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-05T22:45:17.057354Z","title":"Zhang, L","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.06464","last_updated":"2025-08-08T17:13:00Z","snapshot_observed_at":"2026-08-07T13:11:39.868102Z","submitted_at":"2025-08-08T17:13:00Z","title":"Observation of momentum dependent charge density wave gap in EuTe4","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-05T22:45:17.057354Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2508.06464"},"observation_digest":"sha256:be233d12810d2b8ab00bb035e61f6276971bdc01b5dbca29ea4022d6a80af067","observation_id":"99b6e058-53b0-43ab-ba9a-6c69bde88bcc","resolution":{"observed_at":"2026-08-05T22:45:17.057354Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2309.07045","doi":"10.48550/arxiv.2309.07045","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SafetyBench: Evaluating the Safety of Large Language Models with Multiple Choice Questions","venue":"arXiv (Cornell University)","work_id":"e26de3a1-77ec-4538-b0f3-09cea1fc3e90","year":2023},"citing_paper":{"arxiv_id":"2508.06471","last_updated":"2025-08-08T17:21:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-08T17:21:06Z","title":"GLM-4.5: Agentic, Reasoning, and Coding (ARC) Foundation Models","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-11T17:50:08.399160Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2508.06471"},"observation_digest":"sha256:1afe462be9cfdedee697dfe878d4ad33a1c525f9ba5636367c6f53a4b0685147","observation_id":"20151091-1b09-439b-8963-3e7fdbb5a3cc","resolution":{"observed_at":"2026-05-11T17:50:08.558363Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-05T05:29:20.767922Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.05471","last_updated":"2025-09-05T19:57:38Z","snapshot_observed_at":"2026-08-08T16:21:00.454769Z","submitted_at":"2025-09-05T19:57:38Z","title":"Behind the Mask: Benchmarking Camouflaged Jailbreaks in Large Language Models","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-05T05:29:20.767922Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2509.05471"},"observation_digest":"sha256:14978b410cfd4f2626848f94f9b12a87c05d6d291f89edcbb5ae09fcea2b46c9","observation_id":"13039f77-573b-488c-935f-c28a25313167","resolution":{"observed_at":"2026-08-05T05:29:20.767922Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-04T22:33:25.711468Z","title":"Safetybench: Evaluating the safety of large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.07287","last_updated":"2025-09-08T23:44:00Z","snapshot_observed_at":"2026-08-06T22:54:10.353309Z","submitted_at":"2025-09-08T23:44:00Z","title":"Paladin: Defending LLM-enabled Phishing Emails with a New Trigger-Tag Paradigm","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-04T22:33:25.711468Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2509.07287"},"observation_digest":"sha256:82ce839e830315d48a81db0480d17de3e27b242a10482f840390327f20868161","observation_id":"0b4b09e4-fb46-4bf9-9358-e27503e3e903","resolution":{"observed_at":"2026-08-04T22:33:25.711468Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-04T19:55:56.607844Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.08997","last_updated":"2025-09-10T20:47:56Z","snapshot_observed_at":"2026-08-08T06:23:27.336190Z","submitted_at":"2025-09-10T20:47:56Z","title":"YouthSafe: A Youth-Centric Safety Benchmark and Safeguard Model for Large Language Models","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-04T19:55:56.607844Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2509.08997"},"observation_digest":"sha256:827842e596525a303a8895b62846aaa5332473afe0daa74230e19e6ca1c61721","observation_id":"4d32f22f-ce66-49cd-b9fd-ac8d1e2ec432","resolution":{"observed_at":"2026-08-04T19:55:56.607844Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-04T12:45:26.981751Z","title":"Safetybench: Evaluating the safety of large language models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2510.02480","last_updated":"2026-05-27T22:55:59Z","snapshot_observed_at":"2026-08-08T05:52:54.939846Z","submitted_at":"2025-10-02T18:36:10Z","title":"Controlling the Risk of Corrupted Contexts for Language Models via Early-Exiting","version":3},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-04T12:45:26.981751Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2510.02480"},"observation_digest":"sha256:0b1a0b4333589617ca242b17494d6a3807fa923893dc995d483485157cf044b8","observation_id":"31ff1476-4f45-4a69-b5f3-fc965d7af2ae","resolution":{"observed_at":"2026-08-04T12:45:26.981751Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-04T10:40:58.482981Z","title":"I cannot provide advice on this topic","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2510.09330","last_updated":"2026-05-31T10:03:39Z","snapshot_observed_at":"2026-08-06T10:10:13.146102Z","submitted_at":"2025-10-10T12:32:43Z","title":"Safety Game: Inference-Time Alignment of Black-Box LLMs via Constrained Optimization","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-04T10:40:58.482981Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2510.09330"},"observation_digest":"sha256:9e22ea9513a49f067130bcffdd3bec922d43351984e18277b8630c753482939c","observation_id":"29ed8296-e9a6-4df2-a9b4-323305286cc8","resolution":{"observed_at":"2026-08-04T10:40:58.482981Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2309.07045","doi":"10.48550/arxiv.2309.07045","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SafetyBench: Evaluating the Safety of Large Language Models with Multiple Choice Questions","venue":"arXiv (Cornell University)","work_id":"e26de3a1-77ec-4538-b0f3-09cea1fc3e90","year":2023},"citing_paper":{"arxiv_id":"2512.21110","last_updated":"2026-04-24T20:27:34Z","snapshot_observed_at":"2026-07-06T22:39:58.137482Z","submitted_at":"2025-12-24T11:15:57Z","title":"Beyond Context: Large Language Models' Failure to Grasp Users' Intent","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-16T20:09:25.827452Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2512.21110"},"observation_digest":"sha256:eac2704a54e33b38aa8dad1d26573ea4722ed31b038d4611025fc127f912cca3","observation_id":"a7f5c285-c784-4bbb-ace5-834a5f3a7e3e","resolution":{"observed_at":"2026-05-16T20:11:13.678082Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2309.07045","doi":"10.48550/arxiv.2309.07045","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SafetyBench: Evaluating the Safety of Large Language Models with Multiple Choice Questions","venue":"arXiv (Cornell University)","work_id":"e26de3a1-77ec-4538-b0f3-09cea1fc3e90","year":2023},"citing_paper":{"arxiv_id":"2604.02713","last_updated":"2026-04-03T04:10:46Z","snapshot_observed_at":"2026-07-06T22:52:02.162758Z","submitted_at":"2026-04-03T04:10:46Z","title":"Breakdowns in Conversational AI: Interactional Failures in Emotionally and Ethically Sensitive Contexts","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-13T19:46:56.456625Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2604.02713"},"observation_digest":"sha256:0d34ddccff1acd978f8c0e29e2d00bf5305ae8dfa71437072e449aa46a09c0b2","observation_id":"f5fbc10e-70a4-445a-8b2a-2e2e99205d45","resolution":{"observed_at":"2026-05-13T19:48:11.316619Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2309.07045","doi":"10.48550/arxiv.2309.07045","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SafetyBench: Evaluating the Safety of Large Language Models with Multiple Choice Questions","venue":"arXiv (Cornell University)","work_id":"e26de3a1-77ec-4538-b0f3-09cea1fc3e90","year":2023},"citing_paper":{"arxiv_id":"2604.14548","last_updated":"2026-04-20T07:51:45Z","snapshot_observed_at":"2026-07-06T23:02:18.425249Z","submitted_at":"2026-04-16T02:24:59Z","title":"VoxSafeBench: Not Just What Is Said, but Who, How, and Where","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T10:19:28.041282Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2604.14548"},"observation_digest":"sha256:68958463c30576ec7edbc166fb1808da106d6032755ff5a7d54a816afebacd81","observation_id":"c903e576-4032-44b3-b529-1bafc14e20cf","resolution":{"observed_at":"2026-05-10T10:24:22.098823Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2309.07045","doi":"10.48550/arxiv.2309.07045","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SafetyBench: Evaluating the Safety of Large Language Models with Multiple Choice Questions","venue":"arXiv (Cornell University)","work_id":"e26de3a1-77ec-4538-b0f3-09cea1fc3e90","year":2023},"citing_paper":{"arxiv_id":"2604.16659","last_updated":"2026-04-17T19:28:07Z","snapshot_observed_at":"2026-07-06T23:03:57.199731Z","submitted_at":"2026-04-17T19:28:07Z","title":"Benign Fine-Tuning Breaks Safety Alignment in Audio LLMs","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-10T08:01:25.938248Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2604.16659"},"observation_digest":"sha256:348103a4c95f58633134bd58bd62b74c363a56dcfe1ab2c882f8a8fd65c44d2a","observation_id":"f1c6253d-ff1e-4731-a693-4590896ba80f","resolution":{"observed_at":"2026-05-10T08:02:24.878897Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2309.07045","doi":"10.48550/arxiv.2309.07045","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SafetyBench: Evaluating the Safety of Large Language Models with Multiple Choice Questions","venue":"arXiv (Cornell University)","work_id":"e26de3a1-77ec-4538-b0f3-09cea1fc3e90","year":2023},"citing_paper":{"arxiv_id":"2604.24074","last_updated":"2026-04-27T05:59:59Z","snapshot_observed_at":"2026-08-02T04:52:09.584764Z","submitted_at":"2026-04-27T05:59:59Z","title":"How Sensitive Are Safety Benchmarks to Judge Configuration Choices?","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-08T03:41:32.338529Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2604.24074"},"observation_digest":"sha256:d8b03a742cba973cad950d94f26d145be453a51b9fa56f35f680c7024e039fee","observation_id":"d67dfdc2-8794-4a7b-a329-e7920a0585f6","resolution":{"observed_at":"2026-05-11T21:56:35.407538Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2309.07045","doi":"10.48550/arxiv.2309.07045","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SafetyBench: Evaluating the Safety of Large Language Models with Multiple Choice Questions","venue":"arXiv (Cornell University)","work_id":"e26de3a1-77ec-4538-b0f3-09cea1fc3e90","year":2023},"citing_paper":{"arxiv_id":"2605.06652","last_updated":"2026-05-07T17:56:41Z","snapshot_observed_at":"2026-07-06T23:19:05.764765Z","submitted_at":"2026-05-07T17:56:41Z","title":"When No Benchmark Exists: Validating Comparative LLM Safety Scoring Without Ground-Truth Labels","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-05-08T12:07:02.778631Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2605.06652"},"observation_digest":"sha256:a58131fbe1b53bbeffcd0b2bdce85417f7171c68f520b0ff267e7c783c7867d0","observation_id":"210eaf0c-c35b-4cd5-bb8a-6d26961a24bd","resolution":{"observed_at":"2026-05-08T21:39:24.707089Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2309.07045","doi":"10.48550/arxiv.2309.07045","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SafetyBench: Evaluating the Safety of Large Language Models with Multiple Choice Questions","venue":"arXiv (Cornell University)","work_id":"e26de3a1-77ec-4538-b0f3-09cea1fc3e90","year":2023},"citing_paper":{"arxiv_id":"2605.10639","last_updated":"2026-05-11T14:27:39Z","snapshot_observed_at":"2026-08-02T13:54:55.570407Z","submitted_at":"2026-05-11T14:27:39Z","title":"Navigating the Sea of LLM Evaluation: Investigating Bias in Toxicity Benchmarks","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-12T05:28:45.453455Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2605.10639"},"observation_digest":"sha256:c056dcbd436948116bdd76cdb864f4fe2dc1581534f46986b3f6d407f2e6a0f1","observation_id":"e61ae913-d454-4b05-90b8-92a50e6ce9fd","resolution":{"observed_at":"2026-05-12T05:31:23.839851Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2309.07045","doi":"10.48550/arxiv.2309.07045","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SafetyBench: Evaluating the Safety of Large Language Models with Multiple Choice Questions","venue":"arXiv (Cornell University)","work_id":"e26de3a1-77ec-4538-b0f3-09cea1fc3e90","year":2023},"citing_paper":{"arxiv_id":"2605.22643","last_updated":"2026-05-22T14:53:30Z","snapshot_observed_at":"2026-07-06T23:32:59.663926Z","submitted_at":"2026-05-21T15:50:18Z","title":"Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-05-22T05:50:28.114140Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2605.22643"},"observation_digest":"sha256:b70b20ce35a3184890fb99d00fea4ccddba53b9eddb0d39a3c5281d03259291b","observation_id":"d7668e7c-4541-4545-b656-62b918729385","resolution":{"observed_at":"2026-05-22T05:51:08.043818Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2309.07045","doi":"10.48550/arxiv.2309.07045","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SafetyBench: Evaluating the Safety of Large Language Models with Multiple Choice Questions","venue":"arXiv (Cornell University)","work_id":"e26de3a1-77ec-4538-b0f3-09cea1fc3e90","year":2023},"citing_paper":{"arxiv_id":"2605.22643","last_updated":"2026-05-22T14:53:30Z","snapshot_observed_at":"2026-07-06T23:32:59.663926Z","submitted_at":"2026-05-21T15:50:18Z","title":"Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety","version":2},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-05-25T06:05:27.736494Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2605.22643"},"observation_digest":"sha256:8f7203fabcea0a46bccda35d90bca7d0e0519c674794ecc4877af254d1fa7aba","observation_id":"4d77745e-3327-4ef7-8c07-7c5138ffe310","resolution":{"observed_at":"2026-05-25T06:06:42.916232Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-07-12T23:33:25.778345Z","title":"org/abs/2309.07045","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28830","last_updated":"2026-04-10T06:55:07Z","snapshot_observed_at":"2026-08-01T21:08:45.096490Z","submitted_at":"2026-04-10T06:55:07Z","title":"Benchmarking Open-Source Safety Guard Models: A Comprehensive Evaluation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-07-12T23:33:25.778345Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2605.28830"},"observation_digest":"sha256:c07f600b6f9f718924c567813dabb5d3b1150fcf7b786be36689579fc4e4e90c","observation_id":"2c26ebce-5e82-406e-aaa1-46c1469485b6","resolution":{"observed_at":"2026-07-12T23:33:25.778345Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2309.07045","doi":"10.48550/arxiv.2309.07045","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SafetyBench: Evaluating the Safety of Large Language Models with Multiple Choice Questions","venue":"arXiv (Cornell University)","work_id":"e26de3a1-77ec-4538-b0f3-09cea1fc3e90","year":2023},"citing_paper":{"arxiv_id":"2606.20626","last_updated":"2026-05-26T17:35:31Z","snapshot_observed_at":"2026-08-06T08:20:13.528819Z","submitted_at":"2026-05-26T17:35:31Z","title":"Efficient Safety Benchmarking via Item Response Theory","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-07-01T15:51:00.484343Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2606.20626"},"observation_digest":"sha256:b1a0b3bf46e192f0cc03ce466b06cef81c435e3b8b0b3373db1acdf5ae75d0fa","observation_id":"7859f42a-5e2b-46f0-acb9-e58f687c05b2","resolution":{"observed_at":"2026-07-01T15:55:48.988875Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2309.07045","doi":"10.48550/arxiv.2309.07045","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SafetyBench: Evaluating the Safety of Large Language Models with Multiple Choice Questions","venue":"arXiv (Cornell University)","work_id":"e26de3a1-77ec-4538-b0f3-09cea1fc3e90","year":2023},"citing_paper":{"arxiv_id":"2606.27632","last_updated":"2026-06-26T01:12:02Z","snapshot_observed_at":"2026-08-01T14:50:11.591955Z","submitted_at":"2026-06-26T01:12:02Z","title":"Yuvion LLM: An Adversarially-Aware Large Language Model for Content And AI Safety","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-06-29T00:46:03.210076Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2606.27632"},"observation_digest":"sha256:e4cf7f9f05e83860a1953a2439a03159fdcbd54e27af50276ebc78694087c5e9","observation_id":"874e7ef1-95b4-4fd1-a4cd-b738bf512552","resolution":{"observed_at":"2026-06-29T00:52:55.751302Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2309.07045","doi":"10.48550/arxiv.2309.07045","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SafetyBench: Evaluating the Safety of Large Language Models with Multiple Choice Questions","venue":"arXiv (Cornell University)","work_id":"e26de3a1-77ec-4538-b0f3-09cea1fc3e90","year":2023},"citing_paper":{"arxiv_id":"2607.00913","last_updated":"2026-07-01T13:18:21Z","snapshot_observed_at":"2026-08-05T09:20:57.270200Z","submitted_at":"2026-07-01T13:18:21Z","title":"Two AI Metrics Diverged: Will it Make All the Difference?","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-07-02T12:29:24.439779Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2607.00913"},"observation_digest":"sha256:184e09723ceb12e98468a20750a3a50cdabbb6f998ff786f167a92b004b808fc","observation_id":"4114ba8d-56ca-476e-bbb9-665ba29f2bbd","resolution":{"observed_at":"2026-07-02T12:36:56.167767Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-02T02:22:11.309343Z","title":"arXiv preprint arXiv:2309.07045 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.14353","last_updated":"2026-07-15T20:36:10Z","snapshot_observed_at":"2026-08-08T14:53:47.723360Z","submitted_at":"2026-07-15T20:36:10Z","title":"Unsafe at any AUC: Unlearned Lessons from Sociotechnical Disasters for Responsible AI","version":1},"reference_index":176,"source":"arxiv_source","source_observed_at":"2026-08-02T02:22:11.309343Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2607.14353"},"observation_digest":"sha256:00dd2f7cebdb4198a24d9cc5ed4d5057589eaee00622710b3830b67d098561dd","observation_id":"501a2773-fc1d-4157-ae95-cde76bae5975","resolution":{"observed_at":"2026-08-02T02:22:11.309343Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-01T06:54:48.369854Z","title":"SafetyBench: Evaluating the safety of large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.21735","last_updated":"2026-07-30T16:46:31Z","snapshot_observed_at":"2026-08-08T02:28:53.662477Z","submitted_at":"2026-07-23T18:34:12Z","title":"What AI Red-Team Evaluations Can and Cannot Prove","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-01T06:54:48.369854Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2607.21735"},"observation_digest":"sha256:8157622dd67af92e9f884fec54b331ce55724037d1c681a6d383007fda6a9767","observation_id":"35546b27-3658-41db-ad14-ecb340c7aaee","resolution":{"observed_at":"2026-08-01T06:54:48.369854Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-02T08:30:02.855362Z","title":"Safetybench: Evaluating the safety of large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.22671","last_updated":"2026-07-06T17:26:46Z","snapshot_observed_at":"2026-08-06T17:18:06.876668Z","submitted_at":"2026-07-06T17:26:46Z","title":"AIR-BENCH Live: An Evolving Safety Benchmark for Foundation Models","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-02T08:30:02.855362Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2607.22671"},"observation_digest":"sha256:3f72d0f1d5074a703c64be2ab8c52d1ff1d3a1d5cceb23da78e850abaf870bde","observation_id":"e6acd021-14d3-4938-a9b6-830ddaa8c29f","resolution":{"observed_at":"2026-08-02T08:30:02.855362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-03T00:55:22.021201Z","title":"arXiv preprint arXiv:2309.07045 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.28636","last_updated":"2026-05-19T13:56:13Z","snapshot_observed_at":"2026-08-06T00:38:04.006327Z","submitted_at":"2026-05-19T13:56:13Z","title":"Chain-of-Models: Cross-Model Auditing for Bias-Robust LLM Judges","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-03T00:55:22.021201Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2607.28636"},"observation_digest":"sha256:349ba16a2faa540723b4cfee231c4019a3008f794bdf30588ea1e6e12bd02f8c","observation_id":"2b2a3d9a-93ff-40a9-b21e-f596346e3570","resolution":{"observed_at":"2026-08-03T00:55:22.021201Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2309.07045/citation-record","integrity":"/paper/2309.07045/integrity","json":"/paper/2309.07045/citation-record.json","paper":"/paper/2309.07045"},"outbound":[],"paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 35 inbound Pith citation observations for arXiv:2309.07045."}