{"as_of":"2026-08-18T05:59:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:39f01b5bc900131487b106ca8aebe2602be4e30a0bdc110fe5a0e2e38cef5db2","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":55,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":55,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-17T06:30:58.91139+00:00","state":"measured"},{"denominator":55,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":55,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T22:17:07.384006Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T04:09:34.829805Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":"2402.05044","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-07-04T04:09:34.829805Z","title":"Salad-bench: A hierarchical and com- prehensive safety benchmark for large language models","venue":null,"work_id":"d5b5f02c-8f82-442b-b1ae-b5ea4d849585","year":2024},"citing_paper":{"arxiv_id":"2406.18495","last_updated":"2024-12-09T20:21:56Z","snapshot_observed_at":"2026-08-12T20:12:20.465098Z","submitted_at":"2024-06-26T16:58:20Z","title":"WildGuard: Open One-Stop Moderation Tools for Safety Risks, Jailbreaks, and Refusals of LLMs","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-17T16:25:14.744887Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2406.18495"},"observation_digest":"sha256:f37a20cdc7d7747cff6dc858f39431a7b18a37abb6a53d6d3d0bbc2e2bf641cf","observation_id":"0f840343-e05c-4f98-b0ef-1094684dc243","resolution":{"observed_at":"2026-05-17T16:25:14.860999Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":"2402.05044","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-07-04T04:09:34.829805Z","title":"Salad-bench: A hierarchical and com- prehensive safety benchmark for large language models","venue":null,"work_id":"d5b5f02c-8f82-442b-b1ae-b5ea4d849585","year":2024},"citing_paper":{"arxiv_id":"2407.21772","last_updated":"2024-08-04T22:13:39Z","snapshot_observed_at":"2026-08-15T05:00:37.014348Z","submitted_at":"2024-07-31T17:48:14Z","title":"ShieldGemma: Generative AI Content Moderation Based on Gemma","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-20T13:17:39.444002Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2407.21772"},"observation_digest":"sha256:3ddbf546f8afef598566bda3088bcff8cb13d4ed5dbdb420a225c821f6c9b5ba","observation_id":"18fb40ce-f64f-42aa-94c8-84d37dc436fd","resolution":{"observed_at":"2026-05-20T13:17:39.519538Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-12T14:13:29.934618Z","title":"Salad-bench: A hierarchical and comprehensive safety benchmark for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.16736","last_updated":"2024-11-23T12:50:33Z","snapshot_observed_at":"2026-08-14T20:11:11.616910Z","submitted_at":"2024-11-23T12:50:33Z","title":"ChemSafetyBench: Benchmarking LLM Safety on Chemistry Domain","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T14:13:29.934618Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2411.16736"},"observation_digest":"sha256:043ba790bd622e76b0d4b237e50ae20791a34f36e0adb18b4554fd4dc4800202","observation_id":"90d8e3c8-4f2f-4112-b717-4f6ae335f782","resolution":{"observed_at":"2026-08-12T14:13:29.934618Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-12T05:56:46.443739Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.19939","last_updated":"2025-05-17T15:14:14Z","snapshot_observed_at":"2026-08-17T19:34:58.995686Z","submitted_at":"2024-11-29T18:56:37Z","title":"VLSBench: Unveiling Visual Leakage in Multimodal Safety","version":3},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-12T05:56:46.443739Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2411.19939"},"observation_digest":"sha256:2042de0f424a350ec4b63174e7d37f2ca96f1f6fefc16cf112b07898dc98fda6","observation_id":"c9feef76-95fb-417d-a443-fc4d3c4a12a7","resolution":{"observed_at":"2026-08-12T05:56:46.443739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-11T13:11:27.272180Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13435","last_updated":"2024-12-18T02:13:13Z","snapshot_observed_at":"2026-08-18T01:23:09.096141Z","submitted_at":"2024-12-18T02:13:13Z","title":"Lightweight Safety Classification Using Pruned Language Models","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-11T13:11:27.272180Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2412.13435"},"observation_digest":"sha256:966dd6a2427bc0d0eaba525be196453007cf6bee112c4ae8e252b7d4403016c4","observation_id":"48c215e3-41d3-4839-9c02-73e7c764ba4f","resolution":{"observed_at":"2026-08-11T13:11:27.272180Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-11T20:13:07.261539Z","title":"Salad-bench: A hierarchical and comprehensive safety benchmark for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.14186","last_updated":"2024-12-22T08:52:15Z","snapshot_observed_at":"2026-08-13T06:20:56.614613Z","submitted_at":"2024-12-08T14:14:16Z","title":"Towards AI-$45^{\\circ}$ Law: A Roadmap to Trustworthy AGI","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-11T20:13:07.261539Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2412.14186"},"observation_digest":"sha256:d219bfeaadc1499d2808a0663effea926a2281767dba1e58c69faef15f82d080","observation_id":"42ba8dac-baab-4d56-93c4-e2db98710fb7","resolution":{"observed_at":"2026-08-11T20:13:07.261539Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-11T14:04:52.192614Z","title":"Salad-bench: A hierarchical and comprehensive safety benchmark for large language models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.15265","last_updated":"2024-12-23T11:06:56Z","snapshot_observed_at":"2026-08-15T09:51:52.809342Z","submitted_at":"2024-12-17T03:03:44Z","title":"Chinese SafetyQA: A Safety Short-form Factuality Benchmark for Large Language Models","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-11T14:04:52.192614Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2412.15265"},"observation_digest":"sha256:5dff70db50e9f8d8ae25638df1c1c887537fbb656906b86066b7b6791e502d90","observation_id":"73669d92-0844-4c01-b558-5ac258a156ac","resolution":{"observed_at":"2026-08-11T14:04:52.192614Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-11T05:58:47.068771Z","title":"Salad-bench: A hierarchical and comprehensive safety benchmark for large language models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.16974","last_updated":"2024-12-22T11:16:53Z","snapshot_observed_at":"2026-08-15T12:31:56.442643Z","submitted_at":"2024-12-22T11:16:53Z","title":"Cannot or Should Not? Automatic Analysis of Refusal Composition in IFT/RLHF Datasets and Refusal Behavior of Black-Box LLMs","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-11T05:58:47.068771Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2412.16974"},"observation_digest":"sha256:3d2cc30c12bac8beaa048a7c5b6f580999f1812f288f94eabe50d11b43aea379","observation_id":"b679eb1e-a6af-4246-8abd-373faf51d28b","resolution":{"observed_at":"2026-08-11T05:58:47.068771Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-10T22:23:07.633435Z","title":"Salad-bench: A hierarchical and comprehensive safety benchmark for large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.02039","last_updated":"2025-08-01T17:05:21Z","snapshot_observed_at":"2026-08-17T11:43:18.244486Z","submitted_at":"2025-01-03T14:35:32Z","title":"An Investigation into Value Misalignment in LLM-Generated Texts for Cultural Heritage","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-10T22:23:07.633435Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2501.02039"},"observation_digest":"sha256:0a30796e8e2e518468a72cda62d2d6b88df879601121c969c42d1e42e5b078aa","observation_id":"cc2772f1-ecce-42d8-81bf-16bf5208455b","resolution":{"observed_at":"2026-08-10T22:23:07.633435Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-10T20:53:59.732280Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07071","last_updated":"2025-06-02T15:40:42Z","snapshot_observed_at":"2026-08-12T00:45:41.935460Z","submitted_at":"2025-01-13T05:53:56Z","title":"Value Compass Benchmarks: A Platform for Fundamental and Validated Evaluation of LLMs Values","version":3},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-10T20:53:59.732280Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2501.07071"},"observation_digest":"sha256:78307d9dfdd2ad735f1e4a1177669cb65c6ea9fe7da649a5c71a9baba7dedebd","observation_id":"1c6c52ea-5495-427d-b620-6428913a1a76","resolution":{"observed_at":"2026-08-10T20:53:59.732280Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-10T18:31:03.410535Z","title":"Salad-bench: A hierarchical and comprehensive safety benchmark for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13952","last_updated":"2025-02-27T07:51:29Z","snapshot_observed_at":"2026-08-14T02:24:34.909939Z","submitted_at":"2025-01-20T06:35:01Z","title":"The Dual-use Dilemma in LLMs: Do Empowering Ethical Capacities Make a Degraded Utility?","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-10T18:31:03.410535Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2501.13952"},"observation_digest":"sha256:3fa00c848464202884587910123913872b2f3a7a1296722164d82ee9a3e7ff97","observation_id":"87a1caa9-b2e8-40d8-924a-017402eaeef5","resolution":{"observed_at":"2026-08-10T18:31:03.410535Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-10T04:37:16.968365Z","title":"Salad-bench: A hierarchical and comprehensive safety benchmark for large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.17749","last_updated":"2025-01-29T16:36:53Z","snapshot_observed_at":"2026-08-10T04:50:16.966807Z","submitted_at":"2025-01-29T16:36:53Z","title":"Early External Safety Testing of OpenAI's o3-mini: Insights from the Pre-Deployment Evaluation","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-10T04:37:16.968365Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2501.17749"},"observation_digest":"sha256:7da01ef7bb785a37e311dcd75459b7d1b1738258d794550379d3b8f2b6a45737","observation_id":"bb969384-9c6c-4882-b7fe-83ea091a5669","resolution":{"observed_at":"2026-08-10T04:37:16.968365Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-09T23:32:16.795095Z","title":"Salad-bench: A hierarchical and comprehensive safety benchmark for large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.18438","last_updated":"2025-01-31T15:39:00Z","snapshot_observed_at":"2026-08-18T02:12:13.894958Z","submitted_at":"2025-01-30T15:45:56Z","title":"o3-mini vs DeepSeek-R1: Which One is Safer?","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-09T23:32:16.795095Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2501.18438"},"observation_digest":"sha256:ec8873115ec350db2226c8f36e2977cafde33b0088c4d8512902dfb42795aecb","observation_id":"1ca8e0f6-43fb-4a43-981e-d7b41ee7c37a","resolution":{"observed_at":"2026-08-09T23:32:16.795095Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-09T13:14:34.043782Z","title":"SALAD-Bench : A hierarchical and comprehensive safety benchmark for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.02153","last_updated":"2025-02-04T09:31:54Z","snapshot_observed_at":"2026-08-16T16:34:47.536642Z","submitted_at":"2025-02-04T09:31:54Z","title":"Vulnerability Mitigation for Safety-Aligned Language Models via Debiasing","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-09T13:14:34.043782Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2502.02153"},"observation_digest":"sha256:0b555913688d5ce697bb70e4de4d019bd96167ad8a327f9914bccf39a70c2133","observation_id":"3de26fcb-5bcd-45b2-afd1-e7c91484d7ce","resolution":{"observed_at":"2026-08-09T13:14:34.043782Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-08T21:12:23.245210Z","title":"Salad-bench: A hierarchical and comprehensive safety benchmark for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.05240","last_updated":"2025-02-12T14:43:02Z","snapshot_observed_at":"2026-08-16T11:25:04.356690Z","submitted_at":"2025-02-07T12:18:20Z","title":"Survey on AI-Generated Media Detection: From Non-MLLM to MLLM","version":2},"reference_index":193,"source":"pdf_text","source_observed_at":"2026-08-08T21:12:23.245210Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2502.05240"},"observation_digest":"sha256:0297e79b90484d3f14d1005014005c68a391d1ec1c7d75166d4ca08894ac2b5c","observation_id":"7fea5de3-51c3-4359-be01-e5b890660e00","resolution":{"observed_at":"2026-08-08T21:12:23.245210Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-08T21:06:55.943327Z","title":"Salad-bench: A hierarchical and comprehensive safety benchmark for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.05242","last_updated":"2026-05-27T09:11:42Z","snapshot_observed_at":"2026-08-13T15:11:30.438627Z","submitted_at":"2025-02-07T13:25:33Z","title":"Beyond External Monitors: Enhancing Transparency of Large Language Models for Easier Monitoring","version":3},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-08T21:06:55.943327Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2502.05242"},"observation_digest":"sha256:50820f7062d9aea4b2f299547cf803de60203a75eb3adc8271ada11086639590","observation_id":"3b5c0a1c-1092-451e-abaf-5fcbafeda044","resolution":{"observed_at":"2026-08-08T21:06:55.943327Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-15T22:17:07.384006Z","title":"Salad-bench: A hierarchical and comprehensive safety benchmark for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.07610","last_updated":"2025-05-19T14:00:52Z","snapshot_observed_at":"2026-08-17T15:27:01.964497Z","submitted_at":"2025-05-12T14:31:51Z","title":"Concept-Level Explainability for Auditing & Steering LLM Responses","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-15T22:17:07.384006Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2505.07610"},"observation_digest":"sha256:672358a74cd8b474f1566231d81ae0ce43469d4b0227439a3f7521a566e2a308","observation_id":"e76e2382-4f89-44d7-bde0-4974df77e60f","resolution":{"observed_at":"2026-08-15T22:17:07.384006Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-15T21:14:02.359957Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.10494","last_updated":"2025-05-15T16:53:41Z","snapshot_observed_at":"2026-08-15T21:05:50.109992Z","submitted_at":"2025-05-15T16:53:41Z","title":"Can You Really Trust Code Copilots? Evaluating Large Language Models from a Code Security Perspective","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-15T21:14:02.359957Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2505.10494"},"observation_digest":"sha256:cc6e27adf708c46eeebe59c7a16d56b9f1c0f874a9beaee6fc1b2c238a6748c1","observation_id":"fc51a2f1-175b-4da0-8f6a-44ff4eb32ce7","resolution":{"observed_at":"2026-08-15T21:14:02.359957Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-15T21:05:17.450679Z","title":"Salad-bench: A hierarchical and comprehensive safety benchmark for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.11049","last_updated":"2025-05-16T09:46:10Z","snapshot_observed_at":"2026-08-17T03:50:08.292424Z","submitted_at":"2025-05-16T09:46:10Z","title":"GuardReasoner-VL: Safeguarding VLMs via Reinforced Reasoning","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-15T21:05:17.450679Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2505.11049"},"observation_digest":"sha256:9aa1880039a88b05c7882844b42e6fb34832398133fbce915d6ee59f29a2e5b3","observation_id":"99ec4a73-7b09-4444-af12-1e58b91d8d04","resolution":{"observed_at":"2026-08-15T21:05:17.450679Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-07T14:52:28.676227Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-15T09:04:33.753491Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:28.676227Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:04ea0da57e2e11cef55b8c3bea2b69f78f0d0bf39a001d2038221516c20f3fab","observation_id":"7e09b11c-0844-42b0-9106-664da1d3e6ec","resolution":{"observed_at":"2026-08-07T14:52:28.676227Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-07T13:28:25.012173Z","title":"SALAD-Bench: A hierarchical and comprehensive safety benchmark for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21828","last_updated":"2025-05-27T23:29:32Z","snapshot_observed_at":"2026-08-17T18:31:52.954921Z","submitted_at":"2025-05-27T23:29:32Z","title":"SAGE-Eval: Evaluating LLMs for Systematic Generalizations of Safety Facts","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T13:28:25.012173Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2505.21828"},"observation_digest":"sha256:147592a40c66bbb679af1d5ba2e7d4a6386b5fdd990a30110878cb816f0b7d3a","observation_id":"82b7de15-ee32-4e31-a2d7-80d1c29af714","resolution":{"observed_at":"2026-08-07T13:28:25.012173Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-07T05:41:25.431089Z","title":"Salad-bench: A hierarchical and comprehensive safety benchmark for large language models, 2024 b","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07330","last_updated":"2025-06-09T00:11:06Z","snapshot_observed_at":"2026-08-15T18:49:04.034202Z","submitted_at":"2025-06-09T00:11:06Z","title":"JavelinGuard: Low-Cost Transformer Architectures for LLM Security","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-07T05:41:25.431089Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2506.07330"},"observation_digest":"sha256:622fd2cebbf0fbab16035b047c7863b2df01a176215ebc8dda3fda833661ac01","observation_id":"afdd77a0-996b-4739-bb66-7546694a0b4e","resolution":{"observed_at":"2026-08-07T05:41:25.431089Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-07T10:17:26.810690Z","title":"Salad-bench: A hierarchical and comprehensive safety benchmark for large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.11094","last_updated":"2025-10-30T06:22:33Z","snapshot_observed_at":"2026-08-13T12:53:10.459412Z","submitted_at":"2025-06-06T05:50:50Z","title":"The Scales of Justitia: A Comprehensive Survey on Safety Evaluation of LLMs","version":2},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-07T10:17:26.810690Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2506.11094"},"observation_digest":"sha256:4055dfe41198e4037bd713b65da9e311ea48bcf4055b1636bdbd8f81a773c263","observation_id":"d7eb6bf1-2690-4630-8596-899c53113dab","resolution":{"observed_at":"2026-08-07T10:17:26.810690Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-06T23:00:19.566396Z","title":"Salad-bench: A hierarchical and com- prehensive safety benchmark for large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20251","last_updated":"2025-06-25T08:52:22Z","snapshot_observed_at":"2026-08-13T15:59:23.352815Z","submitted_at":"2025-06-25T08:52:22Z","title":"Q-resafe: Assessing Safety Risks and Quantization-aware Safety Patching for Quantized Large Language Models","version":1},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-06T23:00:19.566396Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2506.20251"},"observation_digest":"sha256:4588304d1340b41724c52b3fc8035c4fbd543cfdd9d81bbbe342db1fee2ccdf7","observation_id":"7a90c41f-4898-42d3-86d6-c6a44e90120a","resolution":{"observed_at":"2026-08-06T23:00:19.566396Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-06T15:14:22.874665Z","title":"Salad-bench: A hierarchical and comprehensive safety benchmark for large language models.arXiv preprint arXiv:2402.05044, 2024a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.16534","last_updated":"2025-07-26T12:33:42Z","snapshot_observed_at":"2026-08-15T01:11:58.606591Z","submitted_at":"2025-07-22T12:44:38Z","title":"Frontier AI Risk Management Framework in Practice: A Risk Analysis Technical Report","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:22.874665Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2507.16534"},"observation_digest":"sha256:01b507ed0e153dda9df7f16912ff3a3635490a18149d0438159299d4f8bb0785","observation_id":"f19f1ff8-ae1f-48e5-9ba7-d3826bd5e83a","resolution":{"observed_at":"2026-08-06T15:14:22.874665Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-15T18:13:58.160012Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18631","last_updated":"2025-07-25T07:20:24Z","snapshot_observed_at":"2026-08-15T18:07:02.667074Z","submitted_at":"2025-07-24T17:59:24Z","title":"Layer-Aware Representation Filtering: Purifying Finetuning Data to Preserve LLM Safety Alignment","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-15T18:13:58.160012Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2507.18631"},"observation_digest":"sha256:65c98bb68a0d4d2874207b63295b0827e35c0000c90eeedeb8f607eae035ca2b","observation_id":"1f89ed07-990b-4cfc-8f93-ba7549c2b795","resolution":{"observed_at":"2026-08-15T18:13:58.160012Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-06T14:13:06.208065Z","title":"Salad-bench: A hierarchical and comprehensive safety benchmark for large language models, February 2024 b","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19672","last_updated":"2025-07-25T20:52:58Z","snapshot_observed_at":"2026-08-07T12:02:14.124506Z","submitted_at":"2025-07-25T20:52:58Z","title":"Alignment and Safety in Large Language Models: Safety Mechanisms, Training Paradigms, and Emerging Challenges","version":1},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-08-06T14:13:06.208065Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2507.19672"},"observation_digest":"sha256:ff62b160bef07e07c524ad1b908273c3d573a0e6f3494f3eedba96b09acd797c","observation_id":"eec8030b-3e91-463e-90c1-c70534844788","resolution":{"observed_at":"2026-08-06T14:13:06.208065Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-15T16:51:54.798464Z","title":"Sal- adbench: A hierarchical and comprehensive safety benchmark for large language models, 2024a","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.20333","last_updated":"2025-08-28T00:30:25Z","snapshot_observed_at":"2026-08-15T16:44:24.168196Z","submitted_at":"2025-08-28T00:30:25Z","title":"Poison Once, Refuse Forever: Weaponizing Alignment for Injecting Bias in LLMs","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-15T16:51:54.798464Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2508.20333"},"observation_digest":"sha256:898960d454df90ec5e8b733b3dd54b9a5a902a33fa024d45b97fb59b170a4bcc","observation_id":"d5639a06-d99e-4775-809d-bcfd832a40ca","resolution":{"observed_at":"2026-08-15T16:51:54.798464Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-04T19:55:56.522947Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.08997","last_updated":"2025-09-10T20:47:56Z","snapshot_observed_at":"2026-08-16T17:54:40.016782Z","submitted_at":"2025-09-10T20:47:56Z","title":"YouthSafe: A Youth-Centric Safety Benchmark and Safeguard Model for Large Language Models","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-04T19:55:56.522947Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2509.08997"},"observation_digest":"sha256:4e88851b3ac5615aa503125f72a5fbe1c1a46d34ea20dd6953e8841175b21ddd","observation_id":"f4c98dc4-5aa3-46e1-ac3b-58ba065cbbcb","resolution":{"observed_at":"2026-08-04T19:55:56.522947Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":"2402.05044","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-07-04T04:09:34.829805Z","title":"Salad-bench: A hierarchical and com- prehensive safety benchmark for large language models","venue":null,"work_id":"d5b5f02c-8f82-442b-b1ae-b5ea4d849585","year":2024},"citing_paper":{"arxiv_id":"2511.12710","last_updated":"2026-05-18T06:50:41Z","snapshot_observed_at":"2026-08-16T18:42:09.532465Z","submitted_at":"2025-11-16T17:52:07Z","title":"Evolve the Method, Not the Prompts: Evolutionary Synthesis of Jailbreak Attacks on LLMs","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-21T18:58:53.183734Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2511.12710"},"observation_digest":"sha256:2d3c9bf6535c2b5f848fb9d2c4827bd787c740876123cae074f9be4e5cfda598","observation_id":"972a1b0d-36e4-471e-b410-e679f75519cd","resolution":{"observed_at":"2026-05-21T19:00:30.338878Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":"2402.05044","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-07-04T04:09:34.829805Z","title":"Salad-bench: A hierarchical and com- prehensive safety benchmark for large language models","venue":null,"work_id":"d5b5f02c-8f82-442b-b1ae-b5ea4d849585","year":2024},"citing_paper":{"arxiv_id":"2602.05946","last_updated":"2026-05-11T01:44:43Z","snapshot_observed_at":"2026-08-13T09:24:22.880393Z","submitted_at":"2026-02-05T18:01:52Z","title":"f-GRPO and Beyond: Divergence-Based Reinforcement Learning Algorithms for General LLM Alignment","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-16T06:48:35.723474Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2602.05946"},"observation_digest":"sha256:6ecb9e80424365afc57cae2e04f9ac07854a855c84c6ac26cfebaffa8684aaa1","observation_id":"6a2660ee-d8d2-4ed6-b36d-a678614671f2","resolution":{"observed_at":"2026-05-16T06:50:43.017079Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":"2402.05044","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-07-04T04:09:34.829805Z","title":"Salad-bench: A hierarchical and com- prehensive safety benchmark for large language models","venue":null,"work_id":"d5b5f02c-8f82-442b-b1ae-b5ea4d849585","year":2024},"citing_paper":{"arxiv_id":"2602.07340","last_updated":"2026-05-21T07:39:41Z","snapshot_observed_at":"2026-08-17T16:49:42.651643Z","submitted_at":"2026-02-07T03:46:33Z","title":"Revisiting Robustness for LLM Safety Alignment via Selective Geometry Control","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-22T11:17:03.104902Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2602.07340"},"observation_digest":"sha256:e48a94189c0df40ceff9131e5ea8ba4290f1c6a1fe83bcf6a04ae48b4f009730","observation_id":"7a8368d9-5b16-46fd-b3c4-72ad86130f73","resolution":{"observed_at":"2026-05-22T11:21:29.020678Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-02T23:32:41.600436Z","title":"arXiv preprint arXiv:2402.05044","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.13562","last_updated":"2026-06-29T01:49:38Z","snapshot_observed_at":"2026-08-10T22:40:00.929377Z","submitted_at":"2026-02-14T02:37:36Z","title":"Mitigating the Safety-utility Trade-off in LLM Alignment via Adaptive Safe Context Learning","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-02T23:32:41.600436Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2602.13562"},"observation_digest":"sha256:679a81157401a37d040e2214535a1c7f39508f8214b18bc9649a02da14a51610","observation_id":"2b64827f-1289-4556-bf5f-c59ed996b449","resolution":{"observed_at":"2026-08-02T23:32:41.600436Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":"2402.05044","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-07-04T04:09:34.829805Z","title":"Salad-bench: A hierarchical and com- prehensive safety benchmark for large language models","venue":null,"work_id":"d5b5f02c-8f82-442b-b1ae-b5ea4d849585","year":2024},"citing_paper":{"arxiv_id":"2604.07754","last_updated":"2026-04-09T03:20:29Z","snapshot_observed_at":"2026-07-06T22:57:00.904627Z","submitted_at":"2026-04-09T03:20:29Z","title":"The Art of (Mis)alignment: How Fine-Tuning Methods Effectively Misalign and Realign LLMs in Post-Training","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-10T18:18:56.476698Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2604.07754"},"observation_digest":"sha256:2520664e54efee463f520260ea885ba5d58cd147cbbd9d1526e25d275692d1e9","observation_id":"613023cc-c79d-4550-9c8c-0912cc952b54","resolution":{"observed_at":"2026-05-11T00:45:50.699060Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":"2402.05044","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-07-04T04:09:34.829805Z","title":"Salad-bench: A hierarchical and com- prehensive safety benchmark for large language models","venue":null,"work_id":"d5b5f02c-8f82-442b-b1ae-b5ea4d849585","year":2024},"citing_paper":{"arxiv_id":"2604.18756","last_updated":"2026-04-20T19:00:09Z","snapshot_observed_at":"2026-08-14T07:51:40.546112Z","submitted_at":"2026-04-20T19:00:09Z","title":"Towards Understanding the Robustness of Sparse Autoencoders","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-10T05:44:24.532548Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2604.18756"},"observation_digest":"sha256:c7366d33d5289618c8b61a0fda4b6e51b4717fb8686510db38bedeb10b846847","observation_id":"563312bb-96a9-48ba-9bf2-729dc1470fe8","resolution":{"observed_at":"2026-05-10T05:46:10.249715Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":"2402.05044","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-07-04T04:09:34.829805Z","title":"Salad-bench: A hierarchical and com- prehensive safety benchmark for large language models","venue":null,"work_id":"d5b5f02c-8f82-442b-b1ae-b5ea4d849585","year":2024},"citing_paper":{"arxiv_id":"2604.24074","last_updated":"2026-04-27T05:59:59Z","snapshot_observed_at":"2026-08-02T04:52:09.584764Z","submitted_at":"2026-04-27T05:59:59Z","title":"How Sensitive Are Safety Benchmarks to Judge Configuration Choices?","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-08T03:41:32.338529Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2604.24074"},"observation_digest":"sha256:590c05bb58de607266085cb2975f514fb850a992e82191214eb7e0cd75d5f32c","observation_id":"33c36e59-155a-486b-acc1-7b6a8bfe9746","resolution":{"observed_at":"2026-05-11T21:56:34.962018Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":"2402.05044","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-07-04T04:09:34.829805Z","title":"Salad-bench: A hierarchical and com- prehensive safety benchmark for large language models","venue":null,"work_id":"d5b5f02c-8f82-442b-b1ae-b5ea4d849585","year":2024},"citing_paper":{"arxiv_id":"2605.00245","last_updated":"2026-04-30T21:27:16Z","snapshot_observed_at":"2026-08-13T05:23:05.594378Z","submitted_at":"2026-04-30T21:27:16Z","title":"ARMOR 2025: A Military-Aligned Benchmark for Evaluating Large Language Model Safety Beyond Civilian Contexts","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-09T20:00:56.184891Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2605.00245"},"observation_digest":"sha256:89520abd21e481256a5bc9f07cfd72ee095eb5cf49ee95a47eae037502b63fe2","observation_id":"875a15c0-8eb0-4689-8f14-fe3629542743","resolution":{"observed_at":"2026-05-11T15:26:09.513310Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":"2402.05044","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-07-04T04:09:34.829805Z","title":"Salad-bench: A hierarchical and com- prehensive safety benchmark for large language models","venue":null,"work_id":"d5b5f02c-8f82-442b-b1ae-b5ea4d849585","year":2024},"citing_paper":{"arxiv_id":"2605.03179","last_updated":"2026-05-04T21:42:10Z","snapshot_observed_at":"2026-08-15T04:44:41.490284Z","submitted_at":"2026-05-04T21:42:10Z","title":"A Validated Prompt Bank for Malicious Code Generation: Separating Executable Weapons from Security Knowledge in 1,554 Consensus-Labeled Prompts","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-08T18:11:29.066362Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2605.03179"},"observation_digest":"sha256:e8ddf3eb69892685dbc4bfea6228ff38d93ab86a51ffe37d2134801bc1f41621","observation_id":"2fd35a25-d2b1-4cb9-9fc5-e917b618f618","resolution":{"observed_at":"2026-05-09T06:40:43.434098Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":"2402.05044","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-07-04T04:09:34.829805Z","title":"Salad-bench: A hierarchical and com- prehensive safety benchmark for large language models","venue":null,"work_id":"d5b5f02c-8f82-442b-b1ae-b5ea4d849585","year":2024},"citing_paper":{"arxiv_id":"2605.03226","last_updated":"2026-06-08T16:31:56Z","snapshot_observed_at":"2026-08-17T02:44:14.919973Z","submitted_at":"2026-05-04T23:30:29Z","title":"Self-Mined Hardness for Safety Fine-Tuning","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-05-08T18:23:49.808496Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2605.03226"},"observation_digest":"sha256:65235197f7519307087da691e829caa88376380d39982fb96361b9c673afb05f","observation_id":"f8f3cead-2e08-4785-bc15-28674a222cf0","resolution":{"observed_at":"2026-05-09T06:30:42.600506Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":"2402.05044","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-07-04T04:09:34.829805Z","title":"Salad-bench: A hierarchical and com- prehensive safety benchmark for large language models","venue":null,"work_id":"d5b5f02c-8f82-442b-b1ae-b5ea4d849585","year":2024},"citing_paper":{"arxiv_id":"2605.03226","last_updated":"2026-06-08T16:31:56Z","snapshot_observed_at":"2026-08-17T02:44:14.919973Z","submitted_at":"2026-05-04T23:30:29Z","title":"Self-Mined Hardness for Safety Fine-Tuning","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-30T23:56:36.068713Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2605.03226"},"observation_digest":"sha256:1b738903dbd92f76622657b60981545431964f5b9bacdb689f493013eea339f3","observation_id":"24fe3dff-2570-45cf-8743-48858c046d7b","resolution":{"observed_at":"2026-07-01T00:55:12.268707Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":"2402.05044","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-07-04T04:09:34.829805Z","title":"Salad-bench: A hierarchical and com- prehensive safety benchmark for large language models","venue":null,"work_id":"d5b5f02c-8f82-442b-b1ae-b5ea4d849585","year":2024},"citing_paper":{"arxiv_id":"2605.10639","last_updated":"2026-05-11T14:27:39Z","snapshot_observed_at":"2026-08-15T07:47:05.762692Z","submitted_at":"2026-05-11T14:27:39Z","title":"Navigating the Sea of LLM Evaluation: Investigating Bias in Toxicity Benchmarks","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-12T05:28:45.453455Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2605.10639"},"observation_digest":"sha256:25cea6f203be5cbfcc2dd435bcd62a04c44c3bd65f98d62a285bc870a6d8d667","observation_id":"4a77e13d-ee30-4f87-8d68-1f95871f76cb","resolution":{"observed_at":"2026-05-12T05:31:23.783532Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":"2402.05044","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-07-04T04:09:34.829805Z","title":"Salad-bench: A hierarchical and com- prehensive safety benchmark for large language models","venue":null,"work_id":"d5b5f02c-8f82-442b-b1ae-b5ea4d849585","year":2024},"citing_paper":{"arxiv_id":"2605.14749","last_updated":"2026-05-14T12:14:42Z","snapshot_observed_at":"2026-08-15T06:07:37.419538Z","submitted_at":"2026-05-14T12:14:42Z","title":"Non-linear Interventions on Large Language Models","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-30T20:58:02.551034Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2605.14749"},"observation_digest":"sha256:a32e3a8dfa69762df624b6e55fcca44d5feebdf9f3b5c769f6b31a945f3d922e","observation_id":"8d2de4f9-441f-4db0-b09a-3f54c8cda564","resolution":{"observed_at":"2026-06-30T21:05:04.711140Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":"2402.05044","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-07-04T04:09:34.829805Z","title":"Salad-bench: A hierarchical and com- prehensive safety benchmark for large language models","venue":null,"work_id":"d5b5f02c-8f82-442b-b1ae-b5ea4d849585","year":2024},"citing_paper":{"arxiv_id":"2605.17329","last_updated":"2026-05-17T08:35:38Z","snapshot_observed_at":"2026-08-15T02:24:24.397804Z","submitted_at":"2026-05-17T08:35:38Z","title":"LPG: Balancing Efficiency and Policy Reasoning in Latent Policy Guardrails","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-19T23:42:58.135553Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2605.17329"},"observation_digest":"sha256:ceeb18cbfd9f6b19be35d22de72fe6581f5e708370792a874dbd0364368b890c","observation_id":"563b547d-e32c-48f8-937e-31ab7586f894","resolution":{"observed_at":"2026-05-19T23:43:17.817102Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-07-12T23:33:25.778345Z","title":"Salad-bench: A hierarchical and comprehensive safety benchmark for large language models.arXiv preprint arXiv:2402.05044,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28830","last_updated":"2026-04-10T06:55:07Z","snapshot_observed_at":"2026-08-14T15:40:33.043140Z","submitted_at":"2026-04-10T06:55:07Z","title":"Benchmarking Open-Source Safety Guard Models: A Comprehensive Evaluation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-07-12T23:33:25.778345Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2605.28830"},"observation_digest":"sha256:4324034d763f77e7de9a186a9312c1dab29360052d10bd06ad622b8e22b470f5","observation_id":"e7a36fd7-91d7-48cc-950b-2db424c69864","resolution":{"observed_at":"2026-07-12T23:33:25.778345Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":"2402.05044","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-07-04T04:09:34.829805Z","title":"Salad-bench: A hierarchical and com- prehensive safety benchmark for large language models","venue":null,"work_id":"d5b5f02c-8f82-442b-b1ae-b5ea4d849585","year":2024},"citing_paper":{"arxiv_id":"2605.29068","last_updated":"2026-05-27T20:15:22Z","snapshot_observed_at":"2026-08-12T23:04:58.335023Z","submitted_at":"2026-05-27T20:15:22Z","title":"Robust and Efficient Guardrails with Latent Reasoning","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-29T12:07:16.574111Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2605.29068"},"observation_digest":"sha256:4272dd01053cff3e2673331dee4a19a09a98497887168e7395f314e09a071cbc","observation_id":"3c81ef08-3245-4c8e-8af3-2801beed71bf","resolution":{"observed_at":"2026-06-29T12:13:26.928220Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":"2402.05044","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-07-04T04:09:34.829805Z","title":"Salad-bench: A hierarchical and com- prehensive safety benchmark for large language models","venue":null,"work_id":"d5b5f02c-8f82-442b-b1ae-b5ea4d849585","year":2024},"citing_paper":{"arxiv_id":"2605.29659","last_updated":"2026-05-28T09:21:42Z","snapshot_observed_at":"2026-08-12T14:52:35.122614Z","submitted_at":"2026-05-28T09:21:42Z","title":"Opir: Efficient Multi-Task Safety Classification for Toxicity, Jailbreaks, Hate Speech, and Harmful Content","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-29T09:11:58.843585Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2605.29659"},"observation_digest":"sha256:3b209d2027afae44262f61eaa72c9b0ba6ef380939da38f7d3aef1ac6d07c7b7","observation_id":"89bc2820-8c3c-4ca8-8891-0f95c7c5933b","resolution":{"observed_at":"2026-06-29T09:13:15.996500Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":"2402.05044","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-07-04T04:09:34.829805Z","title":"Salad-bench: A hierarchical and com- prehensive safety benchmark for large language models","venue":null,"work_id":"d5b5f02c-8f82-442b-b1ae-b5ea4d849585","year":2024},"citing_paper":{"arxiv_id":"2606.00160","last_updated":"2026-05-29T09:04:27Z","snapshot_observed_at":"2026-08-16T16:27:59.247689Z","submitted_at":"2026-05-29T09:04:27Z","title":"DataShield: Safety-degrading Data Filtering for LLM Benign Instruction Fine-Tuning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-28T22:09:02.498712Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2606.00160"},"observation_digest":"sha256:0e2b6ec7be5d058b6197b681254a1e1204fd240f314809c4eb131a0fcb4a2dc4","observation_id":"b24e9e2e-7335-463a-91b3-05c20ea5b272","resolution":{"observed_at":"2026-07-01T19:46:10.195488Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":"2402.05044","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-07-04T04:09:34.829805Z","title":"Salad-bench: A hierarchical and com- prehensive safety benchmark for large language models","venue":null,"work_id":"d5b5f02c-8f82-442b-b1ae-b5ea4d849585","year":2024},"citing_paper":{"arxiv_id":"2606.02530","last_updated":"2026-06-01T17:38:12Z","snapshot_observed_at":"2026-08-05T11:28:53.105574Z","submitted_at":"2026-06-01T17:38:12Z","title":"SafeSteer: Localized On-Policy Distillation for Efficient Safety Alignment","version":1},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-06-28T14:39:11.178976Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2606.02530"},"observation_digest":"sha256:b387695c02bfec1a6b598970faddae8a5f217c981980aa3af7be0ba6bfb82a0a","observation_id":"f2b6b5aa-f2cc-4ff8-8adc-704569036dcf","resolution":{"observed_at":"2026-07-01T23:06:20.820913Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":"2402.05044","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-07-04T04:09:34.829805Z","title":"Salad-bench: A hierarchical and com- prehensive safety benchmark for large language models","venue":null,"work_id":"d5b5f02c-8f82-442b-b1ae-b5ea4d849585","year":2024},"citing_paper":{"arxiv_id":"2606.02959","last_updated":"2026-06-01T23:29:58Z","snapshot_observed_at":"2026-08-13T05:09:28.234752Z","submitted_at":"2026-06-01T23:29:58Z","title":"Gate AI: LLM Security Benchmark Evaluation Methodology and Results","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-28T15:05:08.411286Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2606.02959"},"observation_digest":"sha256:62eeee8d3281a38c04dfb55d75520ac7d573eda65cbee79ee7cc6ee69f300cba","observation_id":"31ebc940-86a9-4941-85f2-25aeb1efc436","resolution":{"observed_at":"2026-07-01T22:46:19.208787Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":"2402.05044","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-07-04T04:09:34.829805Z","title":"Salad-bench: A hierarchical and com- prehensive safety benchmark for large language models","venue":null,"work_id":"d5b5f02c-8f82-442b-b1ae-b5ea4d849585","year":2024},"citing_paper":{"arxiv_id":"2606.19887","last_updated":"2026-06-24T07:42:32Z","snapshot_observed_at":"2026-07-06T23:55:05.932014Z","submitted_at":"2026-06-18T07:46:18Z","title":"FinRED: An Expert-Guided Benchmark Generation and Evaluation Framework for Financial LLM Red-Teaming","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-26T17:11:40.088809Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2606.19887"},"observation_digest":"sha256:8847bdedb74d01b6981429b56023e8193e2c4341fdad774a3894c01c05394b36","observation_id":"26ca5bf8-fa3f-48f2-a22f-0e6b9b7fd927","resolution":{"observed_at":"2026-07-04T04:09:34.831892Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":"2402.05044","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-07-04T04:09:34.829805Z","title":"Salad-bench: A hierarchical and com- prehensive safety benchmark for large language models","venue":null,"work_id":"d5b5f02c-8f82-442b-b1ae-b5ea4d849585","year":2024},"citing_paper":{"arxiv_id":"2606.28153","last_updated":"2026-06-29T15:00:34Z","snapshot_observed_at":"2026-08-14T21:18:47.124582Z","submitted_at":"2026-06-26T14:51:16Z","title":"Robust Harmful Features Under Jailbreak Attacks: Mechanistic Evidence from Attention Head Specialization in Large Language Models","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-06-29T03:35:34.594617Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2606.28153"},"observation_digest":"sha256:bee26b34ff3ef978f0fb7cbf544c70ce3d8d669fb69ecec9fa07d9d65433c144","observation_id":"d302259a-e3c1-473d-b052-99ec4d22ba88","resolution":{"observed_at":"2026-07-01T17:35:51.352924Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-02T14:52:12.598561Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.22545","last_updated":"2026-05-06T10:14:30Z","snapshot_observed_at":"2026-08-17T17:09:48.530619Z","submitted_at":"2026-05-06T10:14:30Z","title":"Semalith v1.4: A Calibrated 184M Safety Classifier Achieving State-of-the-Art Prompt-Injection Detection at 44x Fewer Parameters than Llama-Guard-3-8B","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-02T14:52:12.598561Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2607.22545"},"observation_digest":"sha256:67cbd4ad332f83936c7bece54e49a96d46df82899bf3e1ddfb57eaa871a72706","observation_id":"1bd7fe88-6b44-44b1-9e7c-c53df2ebe41c","resolution":{"observed_at":"2026-08-02T14:52:12.598561Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-07-31T09:24:20.719816Z","title":"arXiv preprint arXiv:2402.05044 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.24898","last_updated":"2026-07-27T16:42:36Z","snapshot_observed_at":"2026-08-15T08:02:42.558488Z","submitted_at":"2026-07-27T16:42:36Z","title":"Harm is not Universal: Community-Specific Toxicity Detection is Urgently Needed","version":1},"reference_index":144,"source":"arxiv_source","source_observed_at":"2026-07-31T09:24:20.719816Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2607.24898"},"observation_digest":"sha256:21dd75cd61553687028bc951fc2b48864052a3c4c2466dff1fcc3a6d3dd76937","observation_id":"ef1e49de-c64a-4939-83ff-62027ec1d88c","resolution":{"observed_at":"2026-07-31T09:24:20.719816Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-03T00:55:22.911556Z","title":"arXiv preprint arXiv:2402.05044 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.28636","last_updated":"2026-05-19T13:56:13Z","snapshot_observed_at":"2026-08-17T01:28:35.393539Z","submitted_at":"2026-05-19T13:56:13Z","title":"Chain-of-Models: Cross-Model Auditing for Bias-Robust LLM Judges","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-03T00:55:22.911556Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2607.28636"},"observation_digest":"sha256:9cbf33416a60a19392abccf2e6573e46e4bb38bccce3fff6c00509400b4492d3","observation_id":"fb3ed657-3a93-4523-86b2-2e210f215efb","resolution":{"observed_at":"2026-08-03T00:55:22.911556Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-15T15:09:53.306180Z","title":"arXiv preprint arXiv:2402.05044 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.01805","last_updated":"2026-08-03T07:13:00Z","snapshot_observed_at":"2026-08-17T07:57:11.663737Z","submitted_at":"2026-08-03T07:13:00Z","title":"CockpitHAT: Dependency-Graph-Driven Hierarchical Attribution for Embodied Multi-Agent Cockpits","version":1},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-15T15:09:53.306180Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2608.01805"},"observation_digest":"sha256:582a66cca402e58c605065df7dab6896b82937eb64954ec93f5152e491dcf630","observation_id":"cfab931d-1af5-4d5b-a163-5258ff903397","resolution":{"observed_at":"2026-08-15T15:09:53.306180Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2402.05044/citation-record","integrity":"/paper/2402.05044/integrity","json":"/paper/2402.05044/citation-record.json","paper":"/paper/2402.05044"},"outbound":[],"paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","latest_version":4,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-16T14:20:31.299446Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 55 inbound Pith citation observations for arXiv:2402.05044."}