{"as_of":"2026-08-09T07:27:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:1ff8d3db0f3f52fceab64fb026451f265a15739fb434c9dc3aaa465c5f8b1721","coverage":[{"denominator":60,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":60,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T19:07:35.553049Z","state":"measured"},{"denominator":60,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":60,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2502.08657/citation-record","integrity":"/paper/2502.08657/integrity","json":"/paper/2502.08657/citation-record.json","paper":"/paper/2502.08657"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-08T19:07:35.351848Z","title":"Gpt-4 technical report,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.351848Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:60e13676c496a6b45b139b2a568fd6f1759f4f0b7a0bca667e83242e7554a628","observation_id":"b90455ae-f6db-4334-8b9b-4dff47035811","resolution":{"observed_at":"2026-08-08T19:07:35.351848Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-08T19:07:35.355831Z","title":"Deepseek-v3 technical report,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.355831Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:04aff759fec30cc0c05227b18f24ed5f4298a9af19a30483bebd3aa5aa72df9a","observation_id":"b33bec3d-afa7-4659-a17d-3d8fe19b393a","resolution":{"observed_at":"2026-08-08T19:07:35.355831Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-08T19:07:35.359608Z","title":"Llama: Open and efficient foundation language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.359608Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:a303d162f030f3d0a0001007d1183cb7998b26c63bbadb2264c74811b0dc988d","observation_id":"44078d6b-9dfb-4df5-88da-a17a08d7b3e1","resolution":{"observed_at":"2026-08-08T19:07:35.359608Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.08073","last_updated":"2022-12-15T06:19:23Z","snapshot_observed_at":"2026-08-02T04:53:58.766070Z","submitted_at":"2022-12-15T06:19:23Z","title":"Constitutional AI: Harmlessness from AI Feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.08073","snapshot_observed_at":"2026-08-08T19:07:35.363738Z","title":"Constitutional ai: Harmlessness from ai feedback,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.363738Z"},"links":{"cited_paper":"/paper/2212.08073","citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:c452f1c452d2334173e0ca4a6a323c12ffbdb0a42ed8147b5ba610df7ecf14cf","observation_id":"f07638cb-ec24-41fe-80bb-cdcd8e3a2928","resolution":{"observed_at":"2026-08-08T19:07:35.363738Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:36.118370Z","title":"The ai alignment problem: why it is hard, and where to start,","venue":null,"work_id":"c871bf26-7817-48ae-891b-aed8f8dc87f5","year":2016},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.367673Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:135264daa69eb152dd57b84980da7e3dc4f292e4ef83b2636558119dcdfb360c","observation_id":"34533a29-e8b8-41ee-9f1e-e620a028d73b","resolution":{"observed_at":"2026-08-08T19:07:36.121774Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:36.108399Z","title":"Artificial intelligence, values, and alignment,","venue":null,"work_id":"c1e60d09-17b0-41ee-9b6e-3143e466b1c3","year":2020},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.371153Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:524d0a9fdd3b4728deaf12c0d1cb38d2b21377ba7683b7a6b9266e12266b49f0","observation_id":"0cc7659b-d836-47d5-9b15-275ffbaebbe6","resolution":{"observed_at":"2026-08-08T19:07:36.111997Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:36.097930Z","title":"Self-instruct: Aligning language models with self- generated instructions,","venue":null,"work_id":"4a0bbfa7-4190-470a-a694-d383f1120c5c","year":2023},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.374888Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:4772ee13c2829e6715b08743905c012978775ffb246c1b51e0841bb0a044c9ea","observation_id":"1c90bfd9-124d-4139-9ae0-dca01410a620","resolution":{"observed_at":"2026-08-08T19:07:36.101679Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:36.088166Z","title":"Self-alignment of large language models via monopolylogue-based social scene simulation,","venue":null,"work_id":"84bec8fe-cf6b-4ff9-8176-707d6787cf4f","year":2024},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.378320Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:178cfb40cabfda68228939f846330ad9ff714f1d21c1a299f596638323d23d8a","observation_id":"27b1f92f-53ec-4980-af67-70b609d027eb","resolution":{"observed_at":"2026-08-08T19:07:36.091151Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00557","last_updated":"2025-06-01T15:48:57Z","snapshot_observed_at":"2026-07-06T18:08:13.456457Z","submitted_at":"2024-05-01T15:06:05Z","title":"Mixture of insighTful Experts (MoTE): The Synergy of Thought Chains and Expert Mixtures in Self-Alignment","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00557","snapshot_observed_at":"2026-08-08T19:07:35.381498Z","title":"Mixture of insightful experts (mote): The synergy of thought chains and expert mixtures in self-alignment,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.381498Z"},"links":{"cited_paper":"/paper/2405.00557","citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:06ee3234fb0212e65a51770f5bd8cc8d4ec21a9a9e9590056c1035653ef3df16","observation_id":"648bc372-db72-460b-ba3d-ef72a90a9c22","resolution":{"observed_at":"2026-08-08T19:07:35.381498Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:36.079517Z","title":"Balancing differential privacy and utility: A relevance-based adaptive private fine-tuning framework for language models,","venue":null,"work_id":"3d672d3f-72aa-45d6-99f0-b10df8627e57","year":2025},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.384930Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:f64d50365d94cb7846df64610fa054d59818fcf48147a40d300b586dd9ae715e","observation_id":"d1901f4e-4d05-43f8-9993-53917d1a0576","resolution":{"observed_at":"2026-08-08T19:07:36.082574Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:36.070531Z","title":"Hard adversarial example mining for improving robust fairness,","venue":null,"work_id":"b851ce2e-fc98-4a57-b647-ea220b08a2ab","year":2025},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.388365Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:4220490c2a07cf0707011112fb6cd1efd9b815c199b0ac998606dd46e267d6c5","observation_id":"efcc71b5-2986-463c-80ce-ffc8f2c53bc7","resolution":{"observed_at":"2026-08-08T19:07:36.073695Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:36.061333Z","title":"Finetuned language models are zero-shot learners,","venue":null,"work_id":"e291f51d-927e-4b87-818c-921db7853304","year":2022},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.392020Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:c874e049ecb5f4845b2e8ae357f3fe1c88bd5023f55a81d66b4393ab03ef4a20","observation_id":"76e5cd17-b69e-40a1-b2d1-2e1afcbc0404","resolution":{"observed_at":"2026-08-08T19:07:36.064386Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2204.05862","last_updated":"2022-04-12T15:02:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-12T15:02:38Z","title":"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.05862","snapshot_observed_at":"2026-08-08T19:07:35.395383Z","title":"Training a helpful and harmless assistant with reinforcement learning from human feedback,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.395383Z"},"links":{"cited_paper":"/paper/2204.05862","citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:bbdfe7fff8824672decbb1d466036a717410d2c1130bc43db07d00a062312359","observation_id":"805dd60b-66fe-4299-86d3-15c85574f5ed","resolution":{"observed_at":"2026-08-08T19:07:35.395383Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:36.051357Z","title":"Fine-tuning aligned language models compromises safety, even when users do not intend to!","venue":null,"work_id":"9c3a21bc-0ec4-4c10-a05e-887eeaf226f8","year":2024},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.399198Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:d74ed3bf39928c39e9a0beeedbd14557c89c9918580338611f4b125820c46205","observation_id":"5bde28a1-a72a-4264-bcc1-2f435ef56ae6","resolution":{"observed_at":"2026-08-08T19:07:36.054791Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.403391Z","title":"Stanford alpaca: An instruction-following llama model,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.403391Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:310f1d88d8c2dbe94d2bcdc2f12bda3e88276cf87ce844169ebfd3d719dfffd1","observation_id":"39e23c0a-ca4b-435c-9abf-ce5dfc04f8f5","resolution":{"observed_at":"2026-08-08T19:07:35.403391Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.406934Z","title":"Vicuna: An open-source chatbot impressing gpt-4 with 90%* chatgpt quality,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.406934Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:64934d9898fb2b45e310ab032649ecdc4297034c022effb3cb118efc3f303d88","observation_id":"34156006-5528-4174-af5e-f2ab8bc7a871","resolution":{"observed_at":"2026-08-08T19:07:35.406934Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:36.029881Z","title":"Self-instruct: Aligning language models with self- generated instructions,","venue":null,"work_id":"28edfd9a-e733-4973-abf1-e75dc05eaac0","year":2023},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.410273Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:c3ce29eb2e97532954cb5a76187acaf337e637319fdedc2fd35e35074032a445","observation_id":"00152d79-6299-4ce9-a8f6-718417e69234","resolution":{"observed_at":"2026-08-08T19:07:36.033196Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:36.020432Z","title":"Training language models to follow instructions with human feedback,","venue":null,"work_id":"af2a92f3-484d-47d3-8905-92f3c1dd7bd6","year":2022},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.413493Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:84d151d5a4eb8847ae04130b0ae749ef6285654093238e33f606597f918ea41a","observation_id":"e2c4b06d-7662-4578-8b26-e836036081b2","resolution":{"observed_at":"2026-08-08T19:07:36.023714Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:36.010967Z","title":"Beavertails: Towards improved safety alignment of llm via a human-preference dataset,","venue":null,"work_id":"7bf3787f-93aa-4b71-a137-9641b878d0b4","year":2024},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.417191Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:cda8daced6224398185c2bc76e9145b993d1b0947ab4ddccd527d1e16de504ce","observation_id":"8399b0d9-c134-414d-bce5-e7dddf93432e","resolution":{"observed_at":"2026-08-08T19:07:36.014430Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:36.001187Z","title":"Poisoning language models during instruction tuning,","venue":null,"work_id":"9fecd527-a79b-4c49-980b-e2ca793ff371","year":2023},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.420482Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:78f0b3435d31e85ecce769ed09715f692e666ad5afe19144fc3a0a60112d6f96","observation_id":"1345c631-0cc0-4ecf-be07-6b52b9a6b7f5","resolution":{"observed_at":"2026-08-08T19:07:36.005000Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.991312Z","title":"Openassistant conversations-democratizing large language model align- ment,","venue":null,"work_id":"4e57f3ef-3edf-4b16-85c9-e5f77bc17bff","year":2024},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.423742Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:39f81d454d98fb620db0219aa541acbeebed759ebf0abb636120f03db49e7ea8","observation_id":"fdbce315-085b-4fea-be85-68c1556a2737","resolution":{"observed_at":"2026-08-08T19:07:35.994753Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.981605Z","title":"Principle-driven self-alignment of language models from scratch with minimal human supervision,","venue":null,"work_id":"8932a20d-7b8b-4304-871f-803af2896b35","year":2024},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.427509Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:acaeb5d1ae359ede7b047d0c267e967bed4d64c3a767e499d319352ad265f481","observation_id":"4b048045-507b-4655-9e95-31ed27700854","resolution":{"observed_at":"2026-08-08T19:07:35.985261Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07124","last_updated":"2023-10-09T03:34:01Z","snapshot_observed_at":"2026-07-06T16:18:06.415304Z","submitted_at":"2023-09-13T17:59:09Z","title":"RAIN: Your Language Models Can Align Themselves without Finetuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.07124","snapshot_observed_at":"2026-08-08T19:07:35.430899Z","title":"Rain: Your lan- guage models can align themselves without finetuning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.430899Z"},"links":{"cited_paper":"/paper/2309.07124","citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:431c794b480a235199a48dddf500e14f78e10d95d1de049f30824ad9da201ce3","observation_id":"d1dbda63-7d4d-4e17-a183-ab31d615a3df","resolution":{"observed_at":"2026-08-08T19:07:35.430899Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.00267","last_updated":"2024-09-03T14:01:54Z","snapshot_observed_at":"2026-07-06T16:13:07.384791Z","submitted_at":"2023-09-01T05:53:33Z","title":"RLAIF vs. RLHF: Scaling Reinforcement Learning from Human Feedback with AI Feedback","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.00267","snapshot_observed_at":"2026-08-08T19:07:35.434529Z","title":"Rlaif: Scaling reinforcement learning from human feedback with ai feedback,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.434529Z"},"links":{"cited_paper":"/paper/2309.00267","citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:01ad42361b7c5c2c28122e0ecd7c1bcdc89944d7f56e5a7e00ff1860eeefbd79","observation_id":"0da6d42f-2839-4d31-8723-13b76ff2c4d6","resolution":{"observed_at":"2026-08-08T19:07:35.434529Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.971839Z","title":"Self-concept clarity development across the lifespan,","venue":null,"work_id":"a75774bb-6837-4fd1-bec9-a23d50fc96f5","year":2017},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.438293Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:2dc9d9bfb222cd1d55ee383c380a47aa130af222ea1d6fbaf6deccc3fc8c83dc","observation_id":"721bf31d-5100-4e67-830a-f891a999e833","resolution":{"observed_at":"2026-08-08T19:07:35.975277Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04412","last_updated":"2025-03-04T00:04:24Z","snapshot_observed_at":"2026-08-09T05:49:56.646876Z","submitted_at":"2024-06-06T18:01:02Z","title":"Spread Preference Annotation: Direct Preference Judgment for Efficient LLM Alignment","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04412","snapshot_observed_at":"2026-08-08T19:07:35.441458Z","title":"Aligning large language models with self-generated preference data,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.441458Z"},"links":{"cited_paper":"/paper/2406.04412","citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:2f28d2131dd960a471d8b8ae9b001adcb47a74d6f3cb27c6d939a05bc9c918f9","observation_id":"8cec7ea0-ce83-4d76-89ee-807450d21061","resolution":{"observed_at":"2026-08-08T19:07:35.441458Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06080","last_updated":"2024-01-12T09:46:10Z","snapshot_observed_at":"2026-07-06T17:14:24.413664Z","submitted_at":"2024-01-11T17:56:59Z","title":"Secrets of RLHF in Large Language Models Part II: Reward Modeling","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06080","snapshot_observed_at":"2026-08-08T19:07:35.445137Z","title":"Secrets of rlhf in large language models part ii: Reward modeling,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.445137Z"},"links":{"cited_paper":"/paper/2401.06080","citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:59f425a119acfd2868e3093353c26703e4f2d4cd15044a4edb9b6c6686ff7316","observation_id":"67d604db-332f-4d2e-a985-bce45b67d65c","resolution":{"observed_at":"2026-08-08T19:07:35.445137Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.962516Z","title":"Variational bayesian un- learning,","venue":null,"work_id":"b62c360d-cd67-4518-892f-d706f72d78d3","year":2020},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.448132Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:efa6f2af7cde2d3bf8d5b3dddcbb062458959afdd76ce9772eefb2371c98bbdc","observation_id":"8a8c2ee2-4adb-404e-af5b-88f0a39a02f2","resolution":{"observed_at":"2026-08-08T19:07:35.965834Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.08747","last_updated":"2025-01-05T04:09:00Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-17T02:53:23Z","title":"An Empirical Study of Catastrophic Forgetting in Large Language Models During Continual Fine-tuning","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.08747","snapshot_observed_at":"2026-08-08T19:07:35.450781Z","title":"An empir- ical study of catastrophic forgetting in large language models during continual fine-tuning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.450781Z"},"links":{"cited_paper":"/paper/2308.08747","citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:8d8ff4a779a6d2ba46f5cf5d11eec02546b1f6b363c8855b4756d53de36dae5c","observation_id":"f09253eb-a71a-4520-8631-7bcd66c226f4","resolution":{"observed_at":"2026-08-08T19:07:35.450781Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.18223","last_updated":"2026-03-18T05:34:39Z","snapshot_observed_at":"2026-08-06T23:27:24.356320Z","submitted_at":"2023-03-31T17:28:46Z","title":"A Survey of Large Language Models","version":19},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.18223","snapshot_observed_at":"2026-08-08T19:07:35.453762Z","title":"A survey of large language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.453762Z"},"links":{"cited_paper":"/paper/2303.18223","citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:d517171bff981d3a11764771e5c527764742cc42c4ac740b4f4725ec566940c1","observation_id":"4f879638-9652-4351-baca-4f6a1d159684","resolution":{"observed_at":"2026-08-08T19:07:35.453762Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2108.07258","last_updated":"2022-07-12T23:45:14Z","snapshot_observed_at":"2026-08-02T09:20:40.804790Z","submitted_at":"2021-08-16T17:50:08Z","title":"On the Opportunities and Risks of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.07258","snapshot_observed_at":"2026-08-08T19:07:35.456785Z","title":"On the opportunities and risks of foundation models,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.456785Z"},"links":{"cited_paper":"/paper/2108.07258","citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:c987fe110f7199eed4b6d197a3e48e60a75cea50b0b5c8b24a1c89390f15bcf7","observation_id":"92cdd58e-ad1c-4c04-8397-27b7704acb45","resolution":{"observed_at":"2026-08-08T19:07:35.456785Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.952037Z","title":"Safety-tuned llamas: Lessons from improving the safety of large language models that follow instructions,","venue":null,"work_id":"c7a6b28c-91cf-48bf-b842-1674702a0db3","year":2024},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.460211Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:cfc545234f0c3c3c51fd00ac053eb1937cb1388ed5737e77213009e5d353e034","observation_id":"55eb9564-193f-4d61-aa02-4d7628961252","resolution":{"observed_at":"2026-08-08T19:07:35.955845Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.940274Z","title":"Fine-grained human feedback gives better rewards for language model training,","venue":null,"work_id":"8c9259d9-ee73-4b97-9729-a5fb8a15743c","year":2024},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.463441Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:619cfac39d2b250acacea1ef731ba2e91c351c4f92b4da11252188f55fe51725","observation_id":"f9b95246-da35-47f7-a07a-4b72592d8596","resolution":{"observed_at":"2026-08-08T19:07:35.944549Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.930986Z","title":"Learning to summarize with human feedback,","venue":null,"work_id":"dd0b28de-8161-4b50-9d41-d79697288d7f","year":2020},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.466743Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:c06ef2a9a9e7f628a45341f86ba750dffe8784ee8ec9753ad2187eae96a5f75b","observation_id":"c5f597f0-13b9-422e-87bb-785a3a9c9b4e","resolution":{"observed_at":"2026-08-08T19:07:35.933971Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.922090Z","title":"Rl4f: Generating natural language feedback with reinforcement learning for repairing model outputs,","venue":null,"work_id":"46a8be0c-9d9e-439a-8ed6-429a11efae7b","year":2023},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.470155Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:b9a447a19c1dbb13c77cf0b6137eea9f73eb2106e7445df2a22f5916b6435742","observation_id":"57659805-3e28-4c17-a855-6d5f70f22100","resolution":{"observed_at":"2026-08-08T19:07:35.925355Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.913188Z","title":"Gaining wisdom from setbacks: Aligning large language models via mistake analysis,","venue":null,"work_id":"4415a1e6-6789-4265-8d49-2dd8c16319cc","year":2024},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.473313Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:471d231a6cef5589cbdb7d9f500932dda4676cb630dbef1f815cb9acc65fcf49","observation_id":"77cd4e56-c207-465f-b4b4-0d46f6b51d63","resolution":{"observed_at":"2026-08-08T19:07:35.916201Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.903422Z","title":"Understanding negative samples in instance dis- criminative self-supervised representation learning,","venue":null,"work_id":"0214c758-5920-433c-a19b-3d01716d4cd7","year":2021},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.476777Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:5f5f65973230333a6a58537251f9d2f1fc8ec2083852e010b299954d1bfbd745","observation_id":"91041920-d5a2-45a2-8069-8d8b760dae0f","resolution":{"observed_at":"2026-08-08T19:07:35.907055Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.11651","last_updated":"2024-04-16T11:41:13Z","snapshot_observed_at":"2026-07-06T17:31:50.029722Z","submitted_at":"2024-02-18T17:10:07Z","title":"Learning From Failure: Integrating Negative Examples when Fine-tuning Large Language Models as Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.11651","snapshot_observed_at":"2026-08-08T19:07:35.479808Z","title":"Learning from failure: Integrating negative examples when fine-tuning large language models as agents,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.479808Z"},"links":{"cited_paper":"/paper/2402.11651","citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:8ae566c0a5e2bd69b92b6ce66f616edb5c5d2dd71e078b23e06f796c4445fb18","observation_id":"c4f2e0d2-7b2f-4c36-a474-aa7acc360373","resolution":{"observed_at":"2026-08-08T19:07:35.479808Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.483473Z","title":"What makes for good views for contrastive learning?","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.483473Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:c6997e542f7ba4a6a82b6863c1e2d383ed68af5597aa31064d4d8330195bc553","observation_id":"00f9962e-41d9-4b82-8d17-eebc325c02ee","resolution":{"observed_at":"2026-08-08T19:07:35.483473Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.887795Z","title":"Neural text degeneration with unlikelihood training,","venue":null,"work_id":"449e6377-6163-4910-a7b5-25bea8e0450d","year":2020},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.486507Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:b75c638515dfdffc7098add21500cf3ac1f47a4036f828545b13752e5cec7c6c","observation_id":"6ea4fda9-0edd-4acf-b225-23184ce5552e","resolution":{"observed_at":"2026-08-08T19:07:35.891016Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.878057Z","title":"Openai. gpt-4v(ision) system card,","venue":null,"work_id":"6a9a2a86-4d43-498e-bec2-4cc3f76ce8e4","year":2023},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.489654Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:98d70b6a9b81953c90b4c93a809d9f65758441c3d6fb6cac7b6cd757375e3b0f","observation_id":"4cde7402-0787-42a0-8e7b-cf85aa83ef18","resolution":{"observed_at":"2026-08-08T19:07:35.881487Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-08T19:07:35.492858Z","title":"Llama 2: Open foundation and fine-tuned chat models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.492858Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:2bdc92706210b2de725865fe76a267750a37b87da9e4d9caa7c11c2a4c2b0193","observation_id":"b78a06e6-75c4-4d4a-9fb6-f9ec1518778e","resolution":{"observed_at":"2026-08-08T19:07:35.492858Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.868232Z","title":"Red teaming language models to reduce harms: Methods, scaling behaviors, and lessons learned,","venue":null,"work_id":"efd8711a-7f05-40a3-b01b-9ab9dc2353b5","year":2022},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.496551Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:e06f3f92425cbedae3b9a368b3e6465216e6bbaccd44db2e9c5680f2ef784fc7","observation_id":"2c835252-f39b-495d-b08e-1a973428ef69","resolution":{"observed_at":"2026-08-08T19:07:35.871580Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-08T19:07:35.499871Z","title":"Gemini: a family of highly capable multimodal models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.499871Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:608a779ababa35a8536718364886a8699a3456ca965e55afa3a657909d1f8761","observation_id":"a4bbf00f-7f71-48e1-8023-4e5969b47e7e","resolution":{"observed_at":"2026-08-08T19:07:35.499871Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.503523Z","title":"Chain-of-thought prompting elicits reasoning in large language models,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.503523Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:d8e8beecd5c6e0bd2c8ba5551da192777c2185effe7f4b3eb5bf3ef63c932aca","observation_id":"b9118a71-173d-4d90-94af-3407ca079fd2","resolution":{"observed_at":"2026-08-08T19:07:35.503523Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.852399Z","title":"The wisdom of hindsight makes language models better instruction followers,","venue":null,"work_id":"0ebdba52-5d9f-46d1-8275-9bfc9b7406de","year":2023},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.506641Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:9933327ef739fb7fc36ed15fe39670fbd6ca0237c439586a43eb5cc42683c9bf","observation_id":"23e369c3-388c-4625-afc6-ab1058cce931","resolution":{"observed_at":"2026-08-08T19:07:35.855924Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.842968Z","title":"Koala: A dialogue model for academic research,","venue":null,"work_id":"87b06700-f38d-44dc-8f72-14c0189018b6","year":2023},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.509806Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:b81b514b41c4fc91a3fe7b669a8b825bbbac60d244b3c94b7e45ed38924abb2f","observation_id":"340786a3-7012-4239-9582-dabbf100e490","resolution":{"observed_at":"2026-08-08T19:07:35.846317Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.833315Z","title":"Glm-130b: An open bilingual pre-trained model,","venue":null,"work_id":"b239f935-8e68-495d-b1eb-238b12f499a7","year":2023},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.513267Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:2037a04258471b9a22afe725d471e77f85c70692a21765bb113b746d21884dd5","observation_id":"61f8cb44-e1ea-464a-900f-ba2eac9be164","resolution":{"observed_at":"2026-08-08T19:07:35.836713Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.822575Z","title":"The curious case of neural text degeneration,","venue":null,"work_id":"f28eb4bb-1d62-4bce-adb2-8f30455b2323","year":2019},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.516447Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:d02bf1fd6a7d1cf5b74b21e42db564fdc622eca0ae4a31f2891df76d464acd7f","observation_id":"a4e281db-3fdd-4ff6-ada7-49ab27538497","resolution":{"observed_at":"2026-08-08T19:07:35.826185Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.812377Z","title":"Lora: Low-rank adaptation of large language models,","venue":null,"work_id":"b678dd10-8257-47f8-a3a8-da4d6bac67e2","year":2021},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.519585Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:395146796d9be18c2f934efe46720453cd61824b4cf8e124ff006c3496738f58","observation_id":"0e9c7d08-786e-4f8e-9bac-9a6946ca5f81","resolution":{"observed_at":"2026-08-08T19:07:35.815745Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-08T19:07:35.523090Z","title":"The llama 3 herd of models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.523090Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:020eda9cb8187ba3f1c1c9074fc22532a430275c08e2145b40f14a12c7518dd4","observation_id":"9ef50156-efc3-498c-a623-37771202d5ad","resolution":{"observed_at":"2026-08-08T19:07:35.523090Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.802559Z","title":"Beyond the imitation game: Quantifying and extrapolating the capabilities of language models,","venue":null,"work_id":"e7c03559-ff98-4b94-838e-c21c7315aa6f","year":2023},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.526699Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:ebff1a83d562308e5b9ca16fbf3ca758517b5c2244da186612c7315c60a4dea3","observation_id":"3fdce5e9-8d92-4dc6-9d1d-0bf8844b6519","resolution":{"observed_at":"2026-08-08T19:07:35.805988Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.792619Z","title":"Autodan: Generating stealthy jailbreak prompts on aligned large language models,","venue":null,"work_id":"ad0903ce-318b-4c14-a4b0-c8df84333248","year":2023},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.529960Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:fd0b0397a4391cf44a91197a05f4283c450935fd3d926d70cc49800def5e9d8d","observation_id":"dd711585-734e-43ee-8ba3-4e08b50fd056","resolution":{"observed_at":"2026-08-08T19:07:35.796019Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.782828Z","title":"Harmbench: A standardized evaluation framework for automated red teaming and robust refusal,","venue":null,"work_id":"c8c08dba-c3ba-47ff-a347-01c66733e629","year":2024},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.533362Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:14d478a4b852eafbed10796dcea306b376fb6c8bb9025c1d486b173ba13ee108","observation_id":"cf6c2c09-2ba9-4856-9d5f-4b273dcf9166","resolution":{"observed_at":"2026-08-08T19:07:35.786305Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.773275Z","title":"Truthfulqa: Measuring how models mimic human falsehoods,","venue":null,"work_id":"45d124a4-0a84-4b47-b794-442406100552","year":2022},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.536655Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:4565ec38f6e4dcbcf6f776494f0ac23d79a8749d891123368272460da359b97f","observation_id":"cf95be37-1915-4c0a-8207-3ac6d74ec4e3","resolution":{"observed_at":"2026-08-08T19:07:35.776407Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.539940Z","title":"Measuring massive multitask language understanding,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.539940Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:576b9d5f2febd23f35f66039ec3e3a8c74b72ea1112c58ebd8d8d39a90278999","observation_id":"937b1e5a-7f3e-4358-b548-0f81431ab153","resolution":{"observed_at":"2026-08-08T19:07:35.539940Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.757720Z","title":"Social iqa: Commonsense reasoning about social interactions,","venue":null,"work_id":"176b5b5c-cb0a-4d51-b378-12a3475ec0bc","year":2019},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.543293Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:2245de857c29710647dda75e19ff674c574a0c1d9e99a1f9560a19a1194278dc","observation_id":"b23b4cd2-24f3-4aec-a5ae-6609841d685b","resolution":{"observed_at":"2026-08-08T19:07:35.761578Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.746947Z","title":"Mitigating the alignment tax of rlhf,","venue":null,"work_id":"6b275e8e-5a79-4f30-9f20-81a2b5df96f8","year":2024},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.546791Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:236dc0b2f2d8b836c3304cc4306fd6017fd2df20822ba5fc048eb2a794067899","observation_id":"5534bcea-024b-408b-bde5-a85d0ea447b1","resolution":{"observed_at":"2026-08-08T19:07:35.750924Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T19:07:35.550161Z","title":"Direct preference optimization: Your language model is secretly a reward model,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.550161Z"},"links":{"citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:665a84b1ef6e4143e58e9d3c736cc2958e55a960846e45d084e4b005ad387068","observation_id":"712610c3-5c13-4414-9edf-2fd41067a154","resolution":{"observed_at":"2026-08-08T19:07:35.550161Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01306","last_updated":"2024-11-19T18:12:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-02T10:53:36Z","title":"KTO: Model Alignment as Prospect Theoretic Optimization","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01306","snapshot_observed_at":"2026-08-08T19:07:35.553049Z","title":"Kto: Model alignment as prospect theoretic optimization,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-08T19:07:35.553049Z"},"links":{"cited_paper":"/paper/2402.01306","citing_paper":"/paper/2502.08657"},"observation_digest":"sha256:62c638c34363e46e1f89c21fb8619464a03b62590672f3aca1f83cb8f8fd8c38","observation_id":"1bfb14e8-30c3-4297-bb0d-779945f7f404","resolution":{"observed_at":"2026-08-08T19:07:35.553049Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2502.08657","last_updated":"2025-02-08T09:54:47Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-08T19:01:10.611273Z","submitted_at":"2025-02-08T09:54:47Z","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions"},"reference_resolution":{"displayed":60,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":24,"verified_exact":0,"verified_fuzzy":36},"total_outbound_references":60},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 60 of 60 outbound references and 0 inbound Pith citation observations for arXiv:2502.08657."}