{"as_of":"2026-08-21T15:47:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:78d3ea97e85622b9413d6f837a9819e9763b19c155a886f34619082df0973eb5","coverage":[{"denominator":32,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":32,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T04:09:26.124106Z","state":"measured"},{"denominator":35,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":35,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-21T06:32:19.484+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T23:13:05.224677Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T12:26:57.033442Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.02009","snapshot_observed_at":"2026-08-05T23:13:05.224677Z","title":"arXiv preprint arXiv:2505.02009 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.05775","last_updated":"2026-07-27T20:50:40Z","snapshot_observed_at":"2026-08-19T11:14:16.487238Z","submitted_at":"2025-08-07T18:42:16Z","title":"Guardians and Offenders: A Survey on Harmful Content Generation and Safety Mitigation of LLM","version":3},"reference_index":212,"source":"arxiv_source","source_observed_at":"2026-08-05T23:13:05.224677Z"},"links":{"cited_paper":"/paper/2505.02009","citing_paper":"/paper/2508.05775"},"observation_digest":"sha256:f54ae04dd3d404b312f3ad9c964662bd1e21ce8b43b4baaa37550c12501d2f93","observation_id":"803bed5d-fdd5-462f-86f8-adeff192f10d","resolution":{"observed_at":"2026-08-05T23:13:05.224677Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.02009","snapshot_observed_at":"2026-08-05T14:57:12.878381Z","title":"Towards safer pretraining: Analyzing and filtering harmful content in webscale datasets for responsible llms, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.20766","last_updated":"2025-08-28T13:22:33Z","snapshot_observed_at":"2026-08-16T04:06:07.476971Z","submitted_at":"2025-08-28T13:22:33Z","title":"Turning the Spell Around: Lightweight Alignment Amplification via Rank-One Safety Injection","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-05T14:57:12.878381Z"},"links":{"cited_paper":"/paper/2505.02009","citing_paper":"/paper/2508.20766"},"observation_digest":"sha256:6b0ebe761fe4fc05d57ac04d954ab753bd45e4b5acde4b46d944a4b376899f48","observation_id":"bb034246-6af4-4a59-b1e4-b960e64aae42","resolution":{"observed_at":"2026-08-05T14:57:12.878381Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"cited_work":{"arxiv_id":"2505.02009","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.02009","snapshot_observed_at":"2026-07-02T12:26:57.033442Z","title":"United Nations","venue":null,"work_id":"9fc74b77-761a-46c6-9861-5d8ae35fe2d6","year":1982},"citing_paper":{"arxiv_id":"2606.05936","last_updated":"2026-06-04T09:38:55Z","snapshot_observed_at":"2026-08-17T13:59:15.148032Z","submitted_at":"2026-06-04T09:38:55Z","title":"Epistemic Injustice in Language Models: An Audit of Pretraining Filters and Guardrails","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-28T02:06:31.341519Z"},"links":{"cited_paper":"/paper/2505.02009","citing_paper":"/paper/2606.05936"},"observation_digest":"sha256:79789d98d826974891493e0860d893dd9ce1456761287721fe49b88336d5e11f","observation_id":"ea22ca27-f0da-4ec6-8833-486a85e51208","resolution":{"observed_at":"2026-07-02T12:26:57.034737Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2505.02009/citation-record","integrity":"/paper/2505.02009/integrity","json":"/paper/2505.02009/citation-record.json","paper":"/paper/2505.02009"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2404.14219","last_updated":"2024-08-30T21:17:17Z","snapshot_observed_at":"2026-08-17T03:25:04.404839Z","submitted_at":"2024-04-22T14:32:33Z","title":"Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.14219","snapshot_observed_at":"2026-08-16T04:09:25.966440Z","title":"Phi-3 technical report: A highly capa- ble language model locally on your phone","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:25.966440Z"},"links":{"cited_paper":"/paper/2404.14219","citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:c15d4c8716d71a5abcfc94cc321a6273be5bdd48c1b5737179ca1451d62a892a","observation_id":"d4925574-3d7e-4987-9230-c4838ca40804","resolution":{"observed_at":"2026-08-16T04:09:25.966440Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.12712","last_updated":"2023-04-13T20:41:31Z","snapshot_observed_at":"2026-08-18T22:17:47.865607Z","submitted_at":"2023-03-22T16:51:28Z","title":"Sparks of Artificial General Intelligence: Early experiments with GPT-4","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.12712","snapshot_observed_at":"2026-08-16T04:09:25.994378Z","title":"Sparks of artificial general intelli- gence: Early experiments with gpt-4","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:25.994378Z"},"links":{"cited_paper":"/paper/2303.12712","citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:e6f302143f71b9fea578cc122c3acb432d77f153f358b9ceb4df49f4f379f2fd","observation_id":"05346a7b-52ee-4d55-b419-aa249f7f858e","resolution":{"observed_at":"2026-08-16T04:09:25.994378Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:09:26.654690Z","title":"Illicit darkweb clas- sification via natural-language processing: Classifying illicit content of webpages based on textual information","venue":null,"work_id":"c01be807-aae5-4baa-be69-df431a56d76b","year":2022},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:26.001429Z"},"links":{"citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:ae616397e0f7f4c452e91a1411331e14bcb657d59930088c6c90daf93bbc6fca","observation_id":"3b647bc3-f328-4af7-9a8e-9a8141c1cba4","resolution":{"observed_at":"2026-08-16T04:09:26.660624Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:09:26.635025Z","title":"HateBERT: Retrain- ing BERT for abusive language detection in English","venue":null,"work_id":"1393aaaa-17fe-4ba0-9cb0-99a071c7f189","year":2021},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:26.006613Z"},"links":{"citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:dbabcac2072c023714bf0952a23095705e9818c1a651873247cd514f6f4c0b34","observation_id":"f6130c86-cf74-411b-9289-054db11cb7ac","resolution":{"observed_at":"2026-08-16T04:09:26.641747Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:09:26.597498Z","title":null,"venue":null,"work_id":"cd4d521f-a858-40dd-8f0a-7f0629dc25ef","year":2025},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:26.015939Z"},"links":{"citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:a9bc06777352c0f870b4fb361ad2a6506ba300ad9ce6a6c19a66bd2b728a2d96","observation_id":"345d4b0e-9013-4a49-8489-fb6b75a02e73","resolution":{"observed_at":"2026-08-16T04:09:26.602751Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-16T04:09:26.024807Z","title":"[Dubey et al., 2024] Abhimanyu Dubey, Abhinav Jauhri, Abhinav Pandey, Abhishek Kadian, Ahmad Al-Dahle, Aiesha Letman, Akhil Mathur, Alan Schelten, Amy Yang, Angela Fan, et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:26.024807Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:8402e0c9d764194d4b38e5bf503f9c8be2574962efa134eb33480a466885308d","observation_id":"b5088d15-c80a-4efe-bc58-362ccaa25ea7","resolution":{"observed_at":"2026-08-16T04:09:26.024807Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:09:26.563406Z","title":null,"venue":null,"work_id":"133d63b1-abb6-4f55-9e9e-38a4f01891a4","year":2020},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:26.029462Z"},"links":{"citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:684f226dbbc1d61f853c6c840bf7f8ca6c36973ca9ad2ac823967dc095237ae4","observation_id":"699ffc2a-35f1-4296-8b27-78c90485a0d9","resolution":{"observed_at":"2026-08-16T04:09:26.568895Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:09:26.547569Z","title":"[Gemma, 2024] Riviere et al","venue":null,"work_id":"96010178-4e2c-4ad6-b705-d0b8346de92e","year":2024},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:26.033791Z"},"links":{"citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:f2b502a0337a0f9feb9baa28d5b085193a98d0819c199b681ff5841483ec49ac","observation_id":"a215cf9b-a5b3-4ed5-a531-9849e4c774f4","resolution":{"observed_at":"2026-08-16T04:09:26.552504Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:09:26.038331Z","title":"A survey on automated fact-checking","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:26.038331Z"},"links":{"citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:3353b5e0ddf1a691eff1921910b0981657c08997b7d087ad14537902776d8462","observation_id":"dc7efd2c-c8cd-438f-a6dc-3f25025189ba","resolution":{"observed_at":"2026-08-16T04:09:26.038331Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:09:26.518969Z","title":"Llama guard: Llm-based input-output safeguard for human-ai conversations,","venue":null,"work_id":"b2b8dd3b-d9d6-412b-acd6-3eaca5aaf267","year":2023},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:26.043728Z"},"links":{"citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:7b6f336b13fbd1113f539d78ede604676be97abe4045730547612ceb95f475b9","observation_id":"7742323f-cfe7-45af-b66b-25d55f925ff4","resolution":{"observed_at":"2026-08-16T04:09:26.524585Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:09:26.500721Z","title":"Perplexed by quality: A perplexity-based method for adult and harmful content detection in multilingual heterogeneous web data,","venue":null,"work_id":"8c8fd186-53f4-4bc5-a083-1c6dff90e4ee","year":2022},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:26.048719Z"},"links":{"citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:56c1f444bbfe1ef80b5891aafa4d5bac47e8197d44bc7f1eecffa6525e827dbe","observation_id":"e413571b-7068-4038-84e7-d96ac011f2d8","resolution":{"observed_at":"2026-08-16T04:09:26.506568Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:09:26.479934Z","title":"A pretrainer’s guide to training data: Measuring the effects of data age, domain coverage, quality, & toxicity","venue":null,"work_id":"492f3adf-0a75-4617-9286-d16febd0f1d1","year":2024},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:26.054262Z"},"links":{"citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:d190920784b7fed80df3579e006698e69374ba601e8fa20d05a98642b8759293","observation_id":"d25c268e-e2b4-4d1a-8ebd-aa952b333597","resolution":{"observed_at":"2026-08-16T04:09:26.486439Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:09:26.461798Z","title":"[Loshchilov and Hutter, 2019] Ilya Loshchilov and Frank Hutter","venue":null,"work_id":"88928007-7b81-4090-811b-ebd52892e524","year":2019},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:26.059337Z"},"links":{"citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:622a8a385dc8a5d8631fc77436588f68e727671162b3198b1c2ce12b90ecc828","observation_id":"5a8a4753-8141-4d5b-8b30-b5d5b6844b03","resolution":{"observed_at":"2026-08-16T04:09:26.468220Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:09:26.424994Z","title":"[Markov et al., 2023] Todor Markov, Chong Zhang, Sand- hini Agarwal, Florentine Eloundou Nekoul, Theodore Lee, Steven Adler, Angela Jiang, and Lilian Weng","venue":null,"work_id":"f54265df-826a-40e2-ad4b-fd0d7311670b","year":2023},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:26.069506Z"},"links":{"citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:2e30023df56315d9145e55f2c89661f482aa37217242b8894a2aa9d161d11821","observation_id":"c2b61ccc-d146-422d-9398-fd8766926cba","resolution":{"observed_at":"2026-08-16T04:09:26.432060Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:09:26.407545Z","title":"Llama 3.2: Revolutionizing edge ai and vision with open, customizable models","venue":null,"work_id":"898f778f-2ee3-4521-9698-2f4c1b4d96d2","year":2024},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:26.074840Z"},"links":{"citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:af4158d8166f96bcf87633d3c6bca55167e66a8a3dc3c52676e7fd07a9116187","observation_id":"51e11f62-3248-474f-be49-7c79497feb97","resolution":{"observed_at":"2026-08-16T04:09:26.412569Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:09:26.391350Z","title":"Mistral-7b-v0.3","venue":null,"work_id":"3b0bee5f-295c-46b0-93e6-a48a521d42f5","year":2024},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:26.079677Z"},"links":{"citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:b8220944a0156be019a35a7e39df0296b8032b6468488b6c20b362535608445c","observation_id":"a8065ae7-ba00-41c9-8ae8-fb8d563ae6cd","resolution":{"observed_at":"2026-08-16T04:09:26.396544Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:09:26.374051Z","title":"Ollama - get up and running with large language models","venue":null,"work_id":"b39ae19b-0eed-4c86-aace-3eabd0fa6a2d","year":2024},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:26.084577Z"},"links":{"citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:cf6165ed1218efe4c8d540514f730758e15216ebd999d7c1739c01b8ba915e76","observation_id":"22d0562e-55fb-4037-a5d2-c1152b222359","resolution":{"observed_at":"2026-08-16T04:09:26.379148Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:09:26.355552Z","title":"Data, data everywhere: A guide for pretraining dataset construction","venue":null,"work_id":"d97e64ad-7056-49ef-807a-2e27a96d19cc","year":2024},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:26.089270Z"},"links":{"citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:022902a23feb3aa0c27bb0ce4f6f6399035e229742d6ad0ff6dd70fbcc9524b7","observation_id":"1fc408a3-7f32-4ab1-8109-72b8d66f5fbc","resolution":{"observed_at":"2026-08-16T04:09:26.361071Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:09:26.338053Z","title":"[Penedo et al., 2024] Guilherme Penedo, Hynek Kydl ´ıˇcek, Loubna Ben allal, Anton Lozhkov, Margaret Mitchell, Colin Raffel, Leandro V on Werra, and Thomas Wolf","venue":null,"work_id":"01a911bc-c2a5-4ac9-a668-9ce01d515c6d","year":2024},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:26.094105Z"},"links":{"citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:c7a986465056bbdce02deb91c56d479154c35211f7d99a906ec9f9740256ab9d","observation_id":"4635636d-4916-40c5-bc2a-c5e95dae7338","resolution":{"observed_at":"2026-08-16T04:09:26.343610Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:09:26.318725Z","title":null,"venue":null,"work_id":"bbe267c9-e328-4174-875b-2d8f347953cf","year":2020},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:26.098958Z"},"links":{"citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:aeadffb9ac8677d0dd52e236bf8ca2e2f3863ff043a5f46aef42684205d88fec","observation_id":"d53124c4-172b-4a91-b5ec-749c2587d68c","resolution":{"observed_at":"2026-08-16T04:09:26.324416Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:09:26.300562Z","title":"Common crawl – building an open web-scale crawl using hadoop,","venue":null,"work_id":"8dd532ad-30da-4ec1-a729-fe0cda2e2d16","year":2010},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:26.103868Z"},"links":{"citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:7c92b2d12de1c96ea7bcd16bb57ad15fa6e497445f78aaafd74c9ed5b7c9f1e6","observation_id":"45f3d535-32df-4deb-826b-cef15983f794","resolution":{"observed_at":"2026-08-16T04:09:26.305902Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:09:26.264164Z","title":"[Truic˘a and Apostol, 2023] Ciprian-Octavian Truic ˘a and Elena-Simona Apostol","venue":null,"work_id":"94a45b2f-d36b-4855-adb3-1bb2bfc1d625","year":2023},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:26.113990Z"},"links":{"citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:4a1ea19d2f22a3e93432ac1d10aa2c1d40011ade806104dfe5e56b1744ea095b","observation_id":"5a2cb45f-01e1-43db-bdae-8e61c63d6046","resolution":{"observed_at":"2026-08-16T04:09:26.269867Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:09:26.245648Z","title":"Gomez, Łukasz Kaiser, and Illia Polosukhin","venue":null,"work_id":"771db160-3afb-4f83-917b-0b4b6f99787f","year":2017},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:26.119077Z"},"links":{"citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:6a7e22bb743b561eaf94b97f37756185c2d29b813446f55fcc88cee821711848","observation_id":"4e8a4e84-1bf9-4403-8acb-ee9130a06e32","resolution":{"observed_at":"2026-08-16T04:09:26.250492Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:09:26.282003Z","title":"Problematic webpage identification: A trilogy of hatespeech, search engines and GPT","venue":null,"work_id":"94301dc7-6e50-47c4-8971-23bff15bdc7e","year":2023},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":2010,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:26.108871Z"},"links":{"citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:efb16c1dfb66f87d70e24f04a42cf4cb324b7e3eb03dfb32eb63971752dd2524","observation_id":"0340ade0-f2c9-44f9-ab90-2b20cd62a7fc","resolution":{"observed_at":"2026-08-16T04:09:26.288078Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:09:26.227449Z","title":"[Wei et al., 2022] Jason Wei, Xuezhi Wang, Dale Schuur- mans, Maarten Bosma, Brian Ichter, Fei Xia, Ed H","venue":null,"work_id":"e9b53fe8-dc64-4597-a9db-67d8285ced7c","year":2022},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:26.124106Z"},"links":{"citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:81d78ca012e252bca13c5567966d64affcfd84a0f76d0fd9c2b77190ce39d011","observation_id":"c2cdef70-35aa-41d9-8261-c059ce7e61ac","resolution":{"observed_at":"2026-08-16T04:09:26.233298Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:09:26.444104Z","title":"What’s in the box? an analysis of undesirable content in the Common Crawl corpus","venue":null,"work_id":"3d5671ff-4a7d-44d6-8260-9de5cda80a53","year":2021},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:26.064197Z"},"links":{"citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:ea80d4d65a5a531dc9e3aafb19f93a284db3e742d7b104530ccfa267fa7bde7d","observation_id":"6be51992-5f56-430c-b6b1-998cad748a1d","resolution":{"observed_at":"2026-08-16T04:09:26.449968Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:09:25.988757Z","title":"Language models are few-shot learn- ers","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:25.988757Z"},"links":{"citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:af1dbb3517d2efc7446b8468d4922307ea1e79bd724939b2da67d373aa47d6b8","observation_id":"59f5f5cf-870c-4cf2-858a-e0daee52ffc1","resolution":{"observed_at":"2026-08-16T04:09:25.988757Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:09:26.616303Z","title":null,"venue":null,"work_id":"195a1265-183d-46ab-9237-f1a589a76bb8","year":2020},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:26.011611Z"},"links":{"citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:3141afc3017f3e77572045d73c976b18616951f3d59d538150125cf0a1a3f59b","observation_id":"da30a9a9-46b4-42e5-bc0f-8e9dc8f1b8af","resolution":{"observed_at":"2026-08-16T04:09:26.621617Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-17T09:58:46.058102Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-16T04:09:25.977828Z","title":"Gpt-4 technical report","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:25.977828Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:7b2807761d59eb7a152cbb82581dbd008bb5d1e41f7077adfc3c2fff4a27190e","observation_id":"2686cbdf-9c57-41ab-8cde-05653776f2ab","resolution":{"observed_at":"2026-08-16T04:09:25.977828Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:09:26.685362Z","title":"Peters, and Ar- man Cohan","venue":null,"work_id":"ed7acd3f-4004-4899-8f77-9f31d9d03bf2","year":2020},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:25.983406Z"},"links":{"citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:6028c2710536bd33257a329fb0a90171fc78578913afaecccdca9686902c9c06","observation_id":"ea172620-5d56-45af-95b4-9d8c8b7f8108","resolution":{"observed_at":"2026-08-16T04:09:26.690671Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:09:26.702786Z","title":"Suicidal ideation detection on social me- dia: A review of machine learning methods,","venue":null,"work_id":"aef48012-a100-46f8-94a9-a50174378e26","year":2022},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:25.972649Z"},"links":{"citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:99dc8e99888d981580d3e04238eb8335ff1a06310259cd0846182a4a1d03d262","observation_id":"4faf27d3-69a6-4065-acf3-ca02546ee48e","resolution":{"observed_at":"2026-08-16T04:09:26.708853Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:09:26.580468Z","title":"Documenting large webtext corpora: A case study on the colossal clean crawled corpus","venue":null,"work_id":"f7f6a273-37c3-47e6-949b-0c45385870b9","year":2021},"citing_paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs","version":3},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-16T04:09:26.020508Z"},"links":{"citing_paper":"/paper/2505.02009"},"observation_digest":"sha256:912d65617342a60fac53c5634b8863889b2a24687b5810d9a4a6502be0d8be41","observation_id":"7de96306-a4bf-4e67-bb0a-fed1115cc866","resolution":{"observed_at":"2026-08-16T04:09:26.585711Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.02009","last_updated":"2025-08-12T19:41:44Z","latest_version":3,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-16T04:01:43.111644Z","submitted_at":"2025-05-04T06:37:20Z","title":"Towards Safer Pretraining: Analyzing and Filtering Harmful Content in Webscale datasets for Responsible LLMs"},"reference_resolution":{"displayed":32,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":10,"verified_exact":0,"verified_fuzzy":22},"total_outbound_references":32},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 32 of 32 outbound references and 3 inbound Pith citation observations for arXiv:2505.02009."}