{"as_of":"2026-08-10T15:02:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:2b695f502a8fb56ac951b0e10da624cbd871fa2f4b6319bc865734e486081628","coverage":[{"denominator":58,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":58,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:52:31.262024Z","state":"measured"},{"denominator":58,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":58,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2505.17332/citation-record","integrity":"/paper/2505.17332/integrity","json":"/paper/2505.17332/citation-record.json","paper":"/paper/2505.17332"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:26.322302Z","title":"online\" 'onlinestring :=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:26.322302Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:b846a7062907be2fb9c6f17089207ca3a52a8f72d4145c602d8dddb8fc97a08e","observation_id":"26b1aae7-672a-45e4-b9c2-5039e9325947","resolution":{"observed_at":"2026-08-07T14:52:26.322302Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:26.464178Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:26.464178Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:cfec87219fe880fed105d254cd1031f5059ecf6d65f5c95c1b5f64ebdfa6ac14","observation_id":"7c087a5a-b0ad-4bcf-b766-8d2e8440c8da","resolution":{"observed_at":"2026-08-07T14:52:26.464178Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.14219","last_updated":"2024-08-30T21:17:17Z","snapshot_observed_at":"2026-08-10T14:07:02.234322Z","submitted_at":"2024-04-22T14:32:33Z","title":"Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.14219","snapshot_observed_at":"2026-08-07T14:52:26.615041Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:26.615041Z"},"links":{"cited_paper":"/paper/2404.14219","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:37a5d98c818773b06bd8a268de4f1521af4f0d0c048c2a40e24eb35e3a7ad42e","observation_id":"64a2ae5a-527d-47ca-b002-dc740e031058","resolution":{"observed_at":"2026-08-07T14:52:26.615041Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:32.997264Z","title":null,"venue":null,"work_id":"dffecbf0-cf44-442b-8592-945b6ee96239","year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:26.750767Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:f7cc910fc3b43c6cf705994c367905ec53752415393bd6a1753668927b136982","observation_id":"c2631a25-09ad-4c04-a86b-ff246c00d569","resolution":{"observed_at":"2026-08-07T14:52:33.061472Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19794","last_updated":"2025-06-11T16:24:02Z","snapshot_observed_at":"2026-08-07T22:59:48.143630Z","submitted_at":"2024-12-27T18:47:05Z","title":"MVTamperBench: Evaluating Robustness of Vision-Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19794","snapshot_observed_at":"2026-08-07T14:52:26.868304Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:26.868304Z"},"links":{"cited_paper":"/paper/2412.19794","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:dd57ddba5ba7d9c50ec7cc2245df766fa0f494a92200c7bf2299a190d25679dc","observation_id":"d0965ad8-1162-4095-8b4a-402b8dc9755b","resolution":{"observed_at":"2026-08-07T14:52:26.868304Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:32.861880Z","title":null,"venue":null,"work_id":"411e2849-32f3-4a29-a6ea-5c554f38bf91","year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:26.976950Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:6c1d74309f2f04bf821dcf025b4765ec9ebb5ad5fc92dfc246966374ef2353e9","observation_id":"e895a1df-56ae-4bd4-a19c-9c5d42c6c30f","resolution":{"observed_at":"2026-08-07T14:52:32.928539Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:32.727447Z","title":null,"venue":null,"work_id":"abae5364-e8ba-4733-9040-65987fbd49e2","year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:27.115403Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:768f71f9c49002a27107aa138e9de94edb705db7fef2743d71aba15e000f2b32","observation_id":"eb97f4e2-8ef0-4ace-883d-09589e055212","resolution":{"observed_at":"2026-08-07T14:52:32.801949Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:27.165158Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:27.165158Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:a965825a727630bfd42fd6ef87c27a19c27d4c5c1989e4f71e757db6313629cf","observation_id":"550c9ee5-1b49-4834-93b7-21747f8a4949","resolution":{"observed_at":"2026-08-07T14:52:27.165158Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.03590","last_updated":"2024-11-27T21:15:02Z","snapshot_observed_at":"2026-07-06T20:01:45.826971Z","submitted_at":"2024-11-27T21:15:02Z","title":"Enhancing Document AI Data Generation Through Graph-Based Synthetic Layouts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.03590","snapshot_observed_at":"2026-08-07T14:52:27.217122Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:27.217122Z"},"links":{"cited_paper":"/paper/2412.03590","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:310ba84a53de787332205952c071f20a76611806652177d38285506bce1748ad","observation_id":"6ee05890-3ee7-49f1-ac40-767856edd6ea","resolution":{"observed_at":"2026-08-07T14:52:27.217122Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:27.279376Z","title":"Do, Yan Xu, and Pascale Fung","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:27.279376Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:35c7c03c6a0cec6a8a997c6d5d77967c185a4998fc60ed2017d3b1d6218b9f34","observation_id":"207e7e05-166f-44cc-b4a5-d9f29d35b8d7","resolution":{"observed_at":"2026-08-07T14:52:27.279376Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2005.14165","last_updated":"2020-07-22T19:47:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-05-28T17:29:03Z","title":"Language Models are Few-Shot Learners","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.14165","snapshot_observed_at":"2026-08-07T14:52:27.438818Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:27.438818Z"},"links":{"cited_paper":"/paper/2005.14165","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:5985c755e63456b0679770be5d93f529245e027c67e5d5b8f652ebcf9b5cfa64","observation_id":"30a523f3-53cd-48df-9482-3c14107d1c63","resolution":{"observed_at":"2026-08-07T14:52:27.438818Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.01318","last_updated":"2024-10-31T22:26:40Z","snapshot_observed_at":"2026-08-02T14:59:12.115203Z","submitted_at":"2024-03-28T02:44:02Z","title":"JailbreakBench: An Open Robustness Benchmark for Jailbreaking Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.01318","snapshot_observed_at":"2026-08-07T14:52:27.510965Z","title":"Pappas, Florian Tramer, Hamed Hassani, and Eric Wong","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:27.510965Z"},"links":{"cited_paper":"/paper/2404.01318","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:500e6eb950dcd7209dff5a6df4357c6b9e04135be0567efdb691263bcf5adfd5","observation_id":"a56ae450-2731-4a80-a54a-990f265db275","resolution":{"observed_at":"2026-08-07T14:52:27.510965Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.16135","last_updated":"2025-03-04T07:00:10Z","snapshot_observed_at":"2026-08-06T00:51:21.827894Z","submitted_at":"2024-06-23T15:15:17Z","title":"Crosslingual Capabilities and Knowledge Barriers in Multilingual Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.16135","snapshot_observed_at":"2026-08-07T14:52:27.565303Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:27.565303Z"},"links":{"cited_paper":"/paper/2406.16135","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:a9408f93725bb84f3541c9ab4ae08c21f7f1d935bf1a9e49058c9568aa1ecdd5","observation_id":"f7d9ba58-cc46-45fd-a320-59388953a377","resolution":{"observed_at":"2026-08-07T14:52:27.565303Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:27.646088Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:27.646088Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:ebc10968266964026802f17a57881330bb8ed53ceb07f3e1fa62884cfc44d576","observation_id":"7f04b227-b7e4-436e-b7d4-918c16af380e","resolution":{"observed_at":"2026-08-07T14:52:27.646088Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-07T14:52:27.742920Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:27.742920Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:26efa5fa37fefbc22a39b8f1f35ceaa684adaba7e4809a4bf1df9d4dc7d91567","observation_id":"a93f0125-3504-4efb-a012-be1992f6f87d","resolution":{"observed_at":"2026-08-07T14:52:27.742920Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.10602","last_updated":"2025-04-25T10:53:27Z","snapshot_observed_at":"2026-08-10T11:35:36.462016Z","submitted_at":"2024-06-15T11:31:39Z","title":"Multilingual Large Language Models and Curse of Multilinguality","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.10602","snapshot_observed_at":"2026-08-07T14:52:27.834797Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:27.834797Z"},"links":{"cited_paper":"/paper/2406.10602","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:993fd2fc1230f52da70d09619db803b25afff812ac7ff98efccb5519794bcf66","observation_id":"4354f0a9-961f-4be5-b27d-78afba2db335","resolution":{"observed_at":"2026-08-07T14:52:27.834797Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.09509","last_updated":"2022-07-14T13:04:29Z","snapshot_observed_at":"2026-07-06T12:49:18.486644Z","submitted_at":"2022-03-17T17:57:56Z","title":"ToxiGen: A Large-Scale Machine-Generated Dataset for Adversarial and Implicit Hate Speech Detection","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.09509","snapshot_observed_at":"2026-08-07T14:52:27.948003Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:27.948003Z"},"links":{"cited_paper":"/paper/2203.09509","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:f24d5be68e3852ee988ea3aef0fef34341d38d8f7828faa7ece47dfdcb036c74","observation_id":"556dabd3-900e-4d59-a3f7-865468e7a661","resolution":{"observed_at":"2026-08-07T14:52:27.948003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-03T06:34:42.243765Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-07T14:52:28.092591Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:28.092591Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:a99f9ded75d4d1fb623284aa8842b23cb2aff3ed6bdda7929d8201568c7a1fd3","observation_id":"fe891c49-9c68-4b27-a31d-636d3040cd64","resolution":{"observed_at":"2026-08-07T14:52:28.092591Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06825","last_updated":"2023-10-10T17:54:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T17:54:58Z","title":"Mistral 7B","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06825","snapshot_observed_at":"2026-08-07T14:52:28.202392Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:28.202392Z"},"links":{"cited_paper":"/paper/2310.06825","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:a14e13e393b3d6b4c1de1c2858b60f1bdfb99e7239cf0d289ca858717e3f3b79","observation_id":"289d9c3d-4374-49f1-90c9-abe2b0413ec1","resolution":{"observed_at":"2026-08-07T14:52:28.202392Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.00515","last_updated":"2024-11-10T22:02:27Z","snapshot_observed_at":"2026-07-29T20:40:25.374189Z","submitted_at":"2024-06-01T17:48:15Z","title":"A Survey on Large Language Models for Code Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.00515","snapshot_observed_at":"2026-08-07T14:52:28.304469Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:28.304469Z"},"links":{"cited_paper":"/paper/2406.00515","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:cd706b536789fb778955f9d3f4c19337aa37731d5503cb8e075051ceb07d1f6b","observation_id":"e14fc6fb-c632-433d-b421-a9306e3bc01b","resolution":{"observed_at":"2026-08-07T14:52:28.304469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:28.406960Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:28.406960Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:a7045a46541119fcef8006aec573af699bac0b64d9b107762a4c6f7a8b76a68e","observation_id":"bb0d6a1a-e472-4331-a9bb-dd6f620818a6","resolution":{"observed_at":"2026-08-07T14:52:28.406960Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.04392","last_updated":"2024-09-09T06:25:33Z","snapshot_observed_at":"2026-08-10T13:20:14.204076Z","submitted_at":"2024-04-05T20:31:45Z","title":"Fine-Tuning, Quantization, and LLMs: Navigating Unintended Outcomes","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.04392","snapshot_observed_at":"2026-08-07T14:52:28.533469Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:28.533469Z"},"links":{"cited_paper":"/paper/2404.04392","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:4a89ec648edcde31814adb76e77c5deb84f0806f8abd98908bc96249519e88f2","observation_id":"d95e5fad-f1cd-4af4-9819-486e74c05bc6","resolution":{"observed_at":"2026-08-07T14:52:28.533469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-03T21:40:53.683032Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-07T14:52:28.676227Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:28.676227Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:b88bcdca1eec71d9fc1bd0cb311e5b4e803b7230b1052773e8fbf0b8c937a9e0","observation_id":"7e09b11c-0844-42b0-9106-664da1d3e6ec","resolution":{"observed_at":"2026-08-07T14:52:28.676227Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12599","last_updated":"2024-08-22T17:59:04Z","snapshot_observed_at":"2026-08-09T21:12:10.507853Z","submitted_at":"2024-08-22T17:59:04Z","title":"Controllable Text Generation for Large Language Models: A Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12599","snapshot_observed_at":"2026-08-07T14:52:28.795914Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:28.795914Z"},"links":{"cited_paper":"/paper/2408.12599","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:feed39b7fdce9ae6f7f4b72a44398bc0eed560a9e927f44c791c0ecc3939fd7f","observation_id":"3ffc685f-ab4a-4baa-923c-52ef5ab7142b","resolution":{"observed_at":"2026-08-07T14:52:28.795914Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-08-07T14:52:28.909375Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:28.909375Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:b3b557e4a3ce3210b7dbdf559cd9865f84f6cd0094113f6369f14b53e0c0e9ce","observation_id":"e773cb32-b65a-4db0-8695-1cd4132b6e08","resolution":{"observed_at":"2026-08-07T14:52:28.909375Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.04656","last_updated":"2024-05-07T20:29:48Z","snapshot_observed_at":"2026-08-09T16:09:14.417005Z","submitted_at":"2024-05-07T20:29:48Z","title":"Corporate Communication Companion (CCC): An LLM-empowered Writing Assistant for Workplace Social Media","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.04656","snapshot_observed_at":"2026-08-07T14:52:28.995452Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:28.995452Z"},"links":{"cited_paper":"/paper/2405.04656","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:807812ef7a2e48dabe3ff1b93fabc2ca3ae9dfc6ce6581e283949ef16c773141","observation_id":"8ada4d7c-2f56-49aa-8e40-16cf10badd7a","resolution":{"observed_at":"2026-08-07T14:52:28.995452Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.01181","last_updated":"2024-04-02T01:56:56Z","snapshot_observed_at":"2026-08-10T07:40:35.263746Z","submitted_at":"2023-05-02T03:27:27Z","title":"A Paradigm Shift: The Future of Machine Translation Lies with Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.01181","snapshot_observed_at":"2026-08-07T14:52:29.088283Z","title":"Wong, Siyou Liu, and Longyue Wang","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:29.088283Z"},"links":{"cited_paper":"/paper/2305.01181","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:9a8e68813a885db24d01839993589393c13affa388d7f6a5e133170c60ea263d","observation_id":"da7635fc-2190-4561-b9c3-f2daa71a078a","resolution":{"observed_at":"2026-08-07T14:52:29.088283Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04249","last_updated":"2024-02-27T04:43:08Z","snapshot_observed_at":"2026-07-06T17:26:23.067923Z","submitted_at":"2024-02-06T18:59:08Z","title":"HarmBench: A Standardized Evaluation Framework for Automated Red Teaming and Robust Refusal","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.04249","snapshot_observed_at":"2026-08-07T14:52:29.187055Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:29.187055Z"},"links":{"cited_paper":"/paper/2402.04249","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:89095924efdd4616b41bd5426f350e4a77a77f972ba62b10ba67fd5f41ba656d","observation_id":"e70cb713-e48b-4f6a-8c0b-d533273190ac","resolution":{"observed_at":"2026-08-07T14:52:29.187055Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:32.583945Z","title":null,"venue":null,"work_id":"10833ee9-6829-41a9-bab8-a07012fd2880","year":2023},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:29.298139Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:55e09203750b1dfaea34d68736dc0605aaeb71a18cb40f2215c16c3ff5c3d69c","observation_id":"96c1bc5c-be2d-424d-b62f-1c8a15e68482","resolution":{"observed_at":"2026-08-07T14:52:32.634095Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:29.369006Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:29.369006Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:c1678c11aed44298e96855fcd01c365e8c9e55f83ccba9d6bc986edb43c5ad03","observation_id":"daf4a831-a6f4-4279-b27f-ce3504892117","resolution":{"observed_at":"2026-08-07T14:52:29.369006Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:29.442168Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:29.442168Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:9cff47f419b99e4384037dac3f2caf4e1a01ce4d1e4ce65c7c97f7d7e905269c","observation_id":"af3517cd-6827-4fff-a2d1-0ce3ba1f690d","resolution":{"observed_at":"2026-08-07T14:52:29.442168Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14962","last_updated":"2024-12-23T19:01:23Z","snapshot_observed_at":"2026-08-09T15:12:16.945174Z","submitted_at":"2024-11-22T14:21:18Z","title":"LLM for Barcodes: Generating Diverse Synthetic Data for Identity Documents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14962","snapshot_observed_at":"2026-08-07T14:52:29.517727Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:29.517727Z"},"links":{"cited_paper":"/paper/2411.14962","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:978b36317872a7f225de36a48bfec091c56c3d983c0780938b94e29ccbcfdf20","observation_id":"b52a48f3-7d90-42a4-a53f-80ae4217600b","resolution":{"observed_at":"2026-08-07T14:52:29.517727Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:29.573151Z","title":"Review of reference generation methods in large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:29.573151Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:13d1d54388e0fcc01bcc8dbd4e7a940f55b205deb80d5ade02ea3f617601a2a2","observation_id":"fa05beb8-01b6-4ffc-96c4-a948895bb029","resolution":{"observed_at":"2026-08-07T14:52:29.573151Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:29.624489Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:29.624489Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:a5190f1046eb9d75b1f1031de2aa5c4c5a5d419920ca64756cd7be469c29a511","observation_id":"7090315f-e05e-42f2-8158-447c8d1cf6fd","resolution":{"observed_at":"2026-08-07T14:52:29.624489Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.16977","last_updated":"2025-04-23T17:28:38Z","snapshot_observed_at":"2026-08-09T15:16:53.974903Z","submitted_at":"2025-04-23T17:28:38Z","title":"Tokenization Matters: Improving Zero-Shot NER for Indic Languages","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.16977","snapshot_observed_at":"2026-08-07T14:52:29.703994Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:29.703994Z"},"links":{"cited_paper":"/paper/2504.16977","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:4e43d413cb86bd09de1459a5e1c1da767c824e8eb0adccbfcabb143758d1200b","observation_id":"5b2fa618-4552-4e31-8336-e1f9e5468b8e","resolution":{"observed_at":"2026-08-07T14:52:29.703994Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13108","last_updated":"2025-04-23T17:13:28Z","snapshot_observed_at":"2026-08-07T18:07:36.239752Z","submitted_at":"2025-02-18T18:20:37Z","title":"Clinical QA 2.0: Multi-Task Learning for Answer Extraction and Categorization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13108","snapshot_observed_at":"2026-08-07T14:52:29.883990Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:29.883990Z"},"links":{"cited_paper":"/paper/2502.13108","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:dc9148630a4c84fdbee2431174eaafa1f10afe21a6dc111da6f987b0695c65a1","observation_id":"686ffdf8-d601-414b-ac05-0c6020e20bd6","resolution":{"observed_at":"2026-08-07T14:52:29.883990Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.17759","last_updated":"2024-12-23T18:15:19Z","snapshot_observed_at":"2026-07-06T20:12:18.144353Z","submitted_at":"2024-12-23T18:15:19Z","title":"Survey of Large Multimodal Model Datasets, Application Categories and Taxonomy","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.17759","snapshot_observed_at":"2026-08-07T14:52:29.947591Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:29.947591Z"},"links":{"cited_paper":"/paper/2412.17759","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:06b1fa2a49e963bfdb65015cf713acc93bd9f212be45c150ac640583b94da6f1","observation_id":"c9cdb51a-fff7-4bf0-8352-94db76df3dcc","resolution":{"observed_at":"2026-08-07T14:52:29.947591Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:30.022676Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.022676Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:8e52f3a75df2ea80aed6106186d2aa1b05275fa55fb399008a82283013246814","observation_id":"69935a2f-d505-4a50-b4ac-003d4e18c356","resolution":{"observed_at":"2026-08-07T14:52:30.022676Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.01263","last_updated":"2024-04-01T11:50:35Z","snapshot_observed_at":"2026-08-03T00:58:55.865010Z","submitted_at":"2023-08-02T16:30:40Z","title":"XSTest: A Test Suite for Identifying Exaggerated Safety Behaviours in Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.01263","snapshot_observed_at":"2026-08-07T14:52:30.095909Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.095909Z"},"links":{"cited_paper":"/paper/2308.01263","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:514fc9a38fab82d086bffc291a7af6aa509967887e34ae8ec0fadb881e59aa7c","observation_id":"52b6447c-c493-4b19-89bc-6796518e7df5","resolution":{"observed_at":"2026-08-07T14:52:30.095909Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:30.159695Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.159695Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:5f4193e95b880a53d58ea28990f9130ab364711e50c8fbfccbe581faa9a7718f","observation_id":"5058c140-2e80-4716-b643-cb7382fe309d","resolution":{"observed_at":"2026-08-07T14:52:30.159695Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.13136","last_updated":"2024-01-23T23:12:09Z","snapshot_observed_at":"2026-07-06T17:19:39.669611Z","submitted_at":"2024-01-23T23:12:09Z","title":"The Language Barrier: Dissecting Safety Challenges of LLMs in Multilingual Contexts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.13136","snapshot_observed_at":"2026-08-07T14:52:30.239795Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.239795Z"},"links":{"cited_paper":"/paper/2401.13136","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:200e6066339e14cd6ec6eba584f155fe5772d93991e96266a9581111ec1ece67","observation_id":"2b1e54c6-b92e-48a4-aefe-4140254a9bb5","resolution":{"observed_at":"2026-08-07T14:52:30.239795Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:32.392450Z","title":null,"venue":null,"work_id":"d453b52c-cb07-4882-8de2-78e86ada9bee","year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.305517Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:7ba6c9b283af1b25748dccc4e3885402b74cafea61a5b5de09953da6ed3268dd","observation_id":"4a3b8bd0-cde7-4ee9-a71d-07972626824b","resolution":{"observed_at":"2026-08-07T14:52:32.452987Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.08377","last_updated":"2023-10-09T15:52:30Z","snapshot_observed_at":"2026-08-04T12:11:05.713517Z","submitted_at":"2023-05-15T06:24:45Z","title":"Text Classification via Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.08377","snapshot_observed_at":"2026-08-07T14:52:30.375712Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.375712Z"},"links":{"cited_paper":"/paper/2305.08377","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:6910a7fc7714520439bb1b22f98fbcd7cb0329e657781104bdc20b254aa469dc","observation_id":"6875d0ee-2f5a-41e4-a4c4-65e207abcc1f","resolution":{"observed_at":"2026-08-07T14:52:30.375712Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:30.469185Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.469185Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:734bc7fe78e6e77fa226ca1cdd543e65350a8aa73e2f3ccf061a9124398a9912","observation_id":"7e4a57dd-0d8c-41dd-b295-f35e2e2bdbe4","resolution":{"observed_at":"2026-08-07T14:52:30.469185Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08676","last_updated":"2024-06-24T08:50:22Z","snapshot_observed_at":"2026-07-06T17:59:29.408432Z","submitted_at":"2024-04-06T15:01:47Z","title":"ALERT: A Comprehensive Benchmark for Assessing Large Language Models' Safety through Red Teaming","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.08676","snapshot_observed_at":"2026-08-07T14:52:30.530713Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.530713Z"},"links":{"cited_paper":"/paper/2404.08676","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:b7140c9223068ab1f46bf84d4983e617f61b2d88daab299e5961b3470865eeb9","observation_id":"9ad2ab34-9314-4d9f-966e-fba96dd00264","resolution":{"observed_at":"2026-08-07T14:52:30.530713Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:32.221140Z","title":null,"venue":null,"work_id":"50eb00b4-ca11-4c0b-b9f9-fa3f31e7f0d1","year":2025},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.572397Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:eb189170abe20b4e0fd8a435ff35cff5e4bcf7923f96e7acfc0eff287bac68ba","observation_id":"fff2e369-c818-4d8d-b4a1-3064e957e2bb","resolution":{"observed_at":"2026-08-07T14:52:32.284403Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-07T14:52:30.616707Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.616707Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:7a024bc9d49789ec7a16a7cc4c487e61ce89dcb4e1f2c5845511deeb31eea0ba","observation_id":"8f548a86-a549-49d8-bf54-2da948643e91","resolution":{"observed_at":"2026-08-07T14:52:30.616707Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.00905","last_updated":"2024-06-20T14:15:23Z","snapshot_observed_at":"2026-08-06T23:53:34.059119Z","submitted_at":"2023-10-02T05:23:34Z","title":"All Languages Matter: On the Multilingual Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.00905","snapshot_observed_at":"2026-08-07T14:52:30.662928Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.662928Z"},"links":{"cited_paper":"/paper/2310.00905","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:a3a5173e5b65c38a9d73dcf78862ccf397673b110991fbe63b8566581613e5b9","observation_id":"5661c39f-23e3-44c4-8c81-356c52133435","resolution":{"observed_at":"2026-08-07T14:52:30.662928Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.13387","last_updated":"2023-09-04T01:47:30Z","snapshot_observed_at":"2026-07-06T16:10:23.328130Z","submitted_at":"2023-08-25T14:02:12Z","title":"Do-Not-Answer: A Dataset for Evaluating Safeguards in LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.13387","snapshot_observed_at":"2026-08-07T14:52:30.727155Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.727155Z"},"links":{"cited_paper":"/paper/2308.13387","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:a1d92261f55b3e7d424e7cc37241efb4e69a64a2916172b361bdc2869e2b505e","observation_id":"8dea832e-a657-4a75-9287-1323bcb16dd5","resolution":{"observed_at":"2026-08-07T14:52:30.727155Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.10523","last_updated":"2024-12-07T09:33:20Z","snapshot_observed_at":"2026-08-06T20:13:51.493067Z","submitted_at":"2024-05-17T04:05:05Z","title":"Adaptable and Reliable Text Classification using Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.10523","snapshot_observed_at":"2026-08-07T14:52:30.798154Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.798154Z"},"links":{"cited_paper":"/paper/2405.10523","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:92fb06f22736ef76618fc6b26d015dd48ce9ea26639a89d8bc279e5b2223b67b","observation_id":"a4d6c8c2-935d-472d-a3e9-0023f6223a01","resolution":{"observed_at":"2026-08-07T14:52:30.798154Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:32.071045Z","title":null,"venue":null,"work_id":"7689dfaa-e966-4cce-869b-006d853f2565","year":2025},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.869862Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:e534b5fbad96348cf5447bfc8d1218fd8fb4c7c7776198d7c56baf077e87ad27","observation_id":"c99df6db-c545-4525-8675-69f32fd72396","resolution":{"observed_at":"2026-08-07T14:52:32.143629Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:31.950366Z","title":null,"venue":null,"work_id":"336d2109-484f-4099-8de2-13ab95e24d4d","year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.926313Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:ab1b63ffca140784473d82676ce91917e24110d78cfc7e817c701bcdbee45d5c","observation_id":"fa468d64-aa82-41e4-93b2-5c3567b007d3","resolution":{"observed_at":"2026-08-07T14:52:31.979578Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14598","last_updated":"2025-03-01T21:45:36Z","snapshot_observed_at":"2026-07-06T18:34:29.513732Z","submitted_at":"2024-06-20T17:56:07Z","title":"SORRY-Bench: Systematically Evaluating Large Language Model Safety Refusal","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.14598","snapshot_observed_at":"2026-08-07T14:52:30.965756Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.965756Z"},"links":{"cited_paper":"/paper/2406.14598","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:d3d1fabd61f26c674ae8d73a142e726eb63709470f6f92cfabeb11163316f17d","observation_id":"6d811496-4a3b-4978-864f-92c658110c44","resolution":{"observed_at":"2026-08-07T14:52:30.965756Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:31.008092Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:31.008092Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:096a96834519a494574a738b598e0cefcecb8853cfa8ef1906c33622ccafb98f","observation_id":"4b32d5de-e91b-44ca-bec3-3bf79eb58ad6","resolution":{"observed_at":"2026-08-07T14:52:31.008092Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:31.085302Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:31.085302Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:311923310a9f2ff56c8fd48542b610ec06191810d6d9523cb605349f300d0815","observation_id":"6df9c3e1-659f-4b29-9f90-bbf5a8fe15e0","resolution":{"observed_at":"2026-08-07T14:52:31.085302Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-09T10:42:31.794884Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-07T14:52:31.148227Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:31.148227Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:91ae57384eb7b5e38311fa120ab29cdc698ba8d45433994d4445e268c5c3e5bb","observation_id":"a444735e-47da-4991-9cee-16ece478ec49","resolution":{"observed_at":"2026-08-07T14:52:31.148227Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:31.213008Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:31.213008Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:e01aa0321343754d2c86fa69940f6ac6aac4075057d51eda7eb6a3b4b497ec19","observation_id":"a1c9744b-1e2d-4831-be5f-977c415f43c2","resolution":{"observed_at":"2026-08-07T14:52:31.213008Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15043","last_updated":"2023-12-20T20:48:57Z","snapshot_observed_at":"2026-07-06T15:59:23.019044Z","submitted_at":"2023-07-27T17:49:12Z","title":"Universal and Transferable Adversarial Attacks on Aligned Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15043","snapshot_observed_at":"2026-08-07T14:52:31.262024Z","title":"Zico Kolter, and Matt Fredrikson","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:31.262024Z"},"links":{"cited_paper":"/paper/2307.15043","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:5aede720cb870f8e77a7fb7fdfed07d70b06534f6cab1454aa2f8b2c0d35f00d","observation_id":"419e77c1-3e29-460a-b529-1c910287c98e","resolution":{"observed_at":"2026-08-07T14:52:31.262024Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-09T16:51:28.094742Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use"},"reference_resolution":{"displayed":58,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":58,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":58},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 58 of 58 outbound references and 0 inbound Pith citation observations for arXiv:2505.17332."}