{"as_of":"2026-08-10T02:26:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d2f7d0a62bb788e271dce9f2083cba63b14a6672edcc4d46d3b8519cb5dbe1ad","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":21,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":21,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":21,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":21,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T15:03:47.464904Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T11:09:46.406911Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-09T15:30:16.931351Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":"2403.00409","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-07-04T11:09:46.406911Z","title":"Provably robust dpo: Aligning language models with noisy feed- back","venue":null,"work_id":"79a01a14-9a55-47f3-85eb-38feebf8582b","year":2024},"citing_paper":{"arxiv_id":"2411.10442","last_updated":"2025-04-07T09:09:39Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-11-15T18:59:27Z","title":"Enhancing the Reasoning Ability of Multimodal Large Language Models via Mixed Preference Optimization","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-16T09:16:17.150383Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2411.10442"},"observation_digest":"sha256:c835e79fffde5022912b518dc957e7b594ba4f8ee99d6f229ae8a5bedc2ef998","observation_id":"2a42b6b5-f3ad-4eb1-96b3-e897de916829","resolution":{"observed_at":"2026-05-16T09:16:17.252104Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-09T15:30:16.931351Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":"2403.00409","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-07-04T11:09:46.406911Z","title":"Provably robust dpo: Aligning language models with noisy feed- back","venue":null,"work_id":"79a01a14-9a55-47f3-85eb-38feebf8582b","year":2024},"citing_paper":{"arxiv_id":"2502.06387","last_updated":"2026-04-07T06:33:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-10T12:15:27Z","title":"How Humans Help LLMs: Assessing and Incentivizing Human Preference Annotators","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-05-23T03:56:18.703995Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2502.06387"},"observation_digest":"sha256:21a24d629af60baccbd875d3af663e62b6d62da1b530e682ca23daea4e9a2e90","observation_id":"0c4164c7-dd8d-4467-a398-af8bbeda1ebf","resolution":{"observed_at":"2026-05-23T03:57:29.679899Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-09T15:30:16.931351Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-08-07T18:23:49.795434Z","title":"Provably robust dpo: Aligning language models with noisy feedback","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.10391","last_updated":"2025-02-14T18:59:51Z","snapshot_observed_at":"2026-08-08T01:24:46.892879Z","submitted_at":"2025-02-14T18:59:51Z","title":"MM-RLHF: The Next Step Forward in Multimodal LLM Alignment","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T18:23:49.795434Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2502.10391"},"observation_digest":"sha256:c3631ce06d8d39c0459c9b642d419716e75e9c912d8417d08e9676af06a4dfdd","observation_id":"6e9ef6e6-ad4d-4ddd-ac35-445bbf735d62","resolution":{"observed_at":"2026-08-07T18:23:49.795434Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-09T15:30:16.931351Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-08-07T15:19:08.819245Z","title":"R., Kini, A., and Natarajan, N","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.15694","last_updated":"2025-05-21T16:07:47Z","snapshot_observed_at":"2026-08-09T15:55:39.140815Z","submitted_at":"2025-05-21T16:07:47Z","title":"A Unified Theoretical Analysis of Private and Robust Offline Alignment: from RLHF to DPO","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T15:19:08.819245Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2505.15694"},"observation_digest":"sha256:66bf57596ba811516d2a77cdba268b46c3304257311405a50a42f0e663f90dad","observation_id":"6f6217b9-2b21-4366-b350-b0d52c2f1c40","resolution":{"observed_at":"2026-08-07T15:19:08.819245Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-09T15:30:16.931351Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":"2403.00409","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-07-04T11:09:46.406911Z","title":"Provably robust dpo: Aligning language models with noisy feed- back","venue":null,"work_id":"79a01a14-9a55-47f3-85eb-38feebf8582b","year":2024},"citing_paper":{"arxiv_id":"2505.19134","last_updated":"2026-04-13T23:58:15Z","snapshot_observed_at":"2026-07-06T21:30:04.828669Z","submitted_at":"2025-05-25T13:11:55Z","title":"Incentivizing High-Quality Human Annotations with Golden Questions","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-19T13:41:26.730528Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2505.19134"},"observation_digest":"sha256:bb3fc448d5af8edc9d1be4b8e03329ed4586841d378a360db460b465aed9956e","observation_id":"55493028-8900-4233-ab07-3174d5ea4748","resolution":{"observed_at":"2026-05-19T13:42:19.320740Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-09T15:30:16.931351Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-08-07T13:44:56.673138Z","title":"R., Kini, A., and Natarajan, N","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21395","last_updated":"2025-05-27T16:23:24Z","snapshot_observed_at":"2026-08-07T13:26:21.678395Z","submitted_at":"2025-05-27T16:23:24Z","title":"Square$\\chi$PO: Differentially Private and Robust $\\chi^2$-Preference Optimization in Offline Direct Alignment","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-07T13:44:56.673138Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2505.21395"},"observation_digest":"sha256:11eccaffbfc2d30ae21c45824717041f3977bbde81990fd7d5d01a2fcb04a1a1","observation_id":"c07e7b14-0763-4b65-88e6-5d97b0cb27d7","resolution":{"observed_at":"2026-08-07T13:44:56.673138Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-09T15:30:16.931351Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-08-07T12:27:53.423556Z","title":"Provably robust dpo: Aligning language models with noisy feedback","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24709","last_updated":"2025-05-30T15:30:43Z","snapshot_observed_at":"2026-08-07T12:12:53.183544Z","submitted_at":"2025-05-30T15:30:43Z","title":"On Symmetric Losses for Robust Policy Optimization with Noisy Preferences","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T12:27:53.423556Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2505.24709"},"observation_digest":"sha256:775e4465a0a1659e097e8316ae8e4c16a28ee417fdf9173a2cbe2c82b6b939b5","observation_id":"18383f47-ebb8-49be-82e0-db1deea0be98","resolution":{"observed_at":"2026-08-07T12:27:53.423556Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-09T15:30:16.931351Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-08-06T19:59:31.844690Z","title":"Provably robust dpo: Aligning language models with noisy feedback.arXiv preprint arXiv:2403.00409, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.04136","last_updated":"2026-07-04T19:39:07Z","snapshot_observed_at":"2026-08-08T15:50:06.172618Z","submitted_at":"2025-07-05T19:13:00Z","title":"A Technical Survey of Reinforcement Learning Techniques for Large Language Models","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-06T19:59:31.844690Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2507.04136"},"observation_digest":"sha256:19afcdca73ed6fc39e21d3d47ff0e0629789201f6ca543eb8794a0662f8d650d","observation_id":"e72950c1-3456-4bd4-8789-0c0fb597fb4a","resolution":{"observed_at":"2026-08-06T19:59:31.844690Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-09T15:30:16.931351Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":"2403.00409","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-07-04T11:09:46.406911Z","title":"Provably robust dpo: Aligning language models with noisy feed- back","venue":null,"work_id":"79a01a14-9a55-47f3-85eb-38feebf8582b","year":2024},"citing_paper":{"arxiv_id":"2509.08933","last_updated":"2026-05-21T17:37:36Z","snapshot_observed_at":"2026-07-06T22:28:28.274132Z","submitted_at":"2025-09-10T18:56:39Z","title":"Corruption-Tolerant Asynchronous Q-Learning with Near-Optimal Rates","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-22T12:58:26.626172Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2509.08933"},"observation_digest":"sha256:335e51e46e7c169c4f5b8bbd3d30addf4fd8b27715a67b7f8e088db80b94602b","observation_id":"879ddf9d-ee39-4e6d-bd23-4b8d4c93c977","resolution":{"observed_at":"2026-05-22T13:01:34.180564Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-09T15:30:16.931351Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-08-04T11:27:22.570045Z","title":"Provably robust dpo: Aligning language models with noisy feedback","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2510.05342","last_updated":"2026-06-01T10:47:12Z","snapshot_observed_at":"2026-08-06T09:31:11.235195Z","submitted_at":"2025-10-06T20:09:37Z","title":"Margin Adaptive DPO: Leveraging Reward Model for Granular Control in Preference Optimization","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-04T11:27:22.570045Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2510.05342"},"observation_digest":"sha256:8bda5422faa7e1903018a49f4523a76582841b4a2130b0a880ac1ea965998060","observation_id":"a0f62a07-09a5-43fe-8a7f-b05f5f17f77f","resolution":{"observed_at":"2026-08-04T11:27:22.570045Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-09T15:30:16.931351Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":"2403.00409","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-07-04T11:09:46.406911Z","title":"Provably robust dpo: Aligning language models with noisy feed- back","venue":null,"work_id":"79a01a14-9a55-47f3-85eb-38feebf8582b","year":2024},"citing_paper":{"arxiv_id":"2510.13830","last_updated":"2026-05-09T20:37:45Z","snapshot_observed_at":"2026-08-08T06:12:13.417479Z","submitted_at":"2025-10-10T08:57:34Z","title":"Users as Annotators: LLM Preference Learning from Comparison Mode","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-18T08:19:58.093621Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2510.13830"},"observation_digest":"sha256:8398009ae993437d05f621c3602aa151283ba79bd04e5ebd49dca67a73ad81c8","observation_id":"b1db7cef-8d31-405d-91c7-233f62db05e9","resolution":{"observed_at":"2026-05-18T08:21:06.992814Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-09T15:30:16.931351Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":"2403.00409","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-07-04T11:09:46.406911Z","title":"Provably robust dpo: Aligning language models with noisy feed- back","venue":null,"work_id":"79a01a14-9a55-47f3-85eb-38feebf8582b","year":2024},"citing_paper":{"arxiv_id":"2605.02971","last_updated":"2026-05-07T19:25:15Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-05-03T14:22:49Z","title":"Multilingual Safety Alignment via Self-Distillation","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-05-08T19:35:56.059362Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2605.02971"},"observation_digest":"sha256:3900b3879456ed9b65bdd5c76ab6f85936d25526689d3bbee327dfc23020aa62","observation_id":"1f275270-dd03-4b92-9031-d2d2159fd45e","resolution":{"observed_at":"2026-05-09T05:45:22.410770Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-09T15:30:16.931351Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":"2403.00409","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-07-04T11:09:46.406911Z","title":"Provably robust dpo: Aligning language models with noisy feed- back","venue":null,"work_id":"79a01a14-9a55-47f3-85eb-38feebf8582b","year":2024},"citing_paper":{"arxiv_id":"2605.02971","last_updated":"2026-05-07T19:25:15Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-05-03T14:22:49Z","title":"Multilingual Safety Alignment via Self-Distillation","version":2},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-05-11T01:08:32.264867Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2605.02971"},"observation_digest":"sha256:1af1602409eba9ca2f11fbac30009d88462c315b44d4e34f1961b74d3d6e9200","observation_id":"10a79da6-4279-4b68-9ab8-028dec443f34","resolution":{"observed_at":"2026-05-11T04:41:00.754016Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-09T15:30:16.931351Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":"2403.00409","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-07-04T11:09:46.406911Z","title":"Provably robust dpo: Aligning language models with noisy feed- back","venue":null,"work_id":"79a01a14-9a55-47f3-85eb-38feebf8582b","year":2024},"citing_paper":{"arxiv_id":"2605.11134","last_updated":"2026-05-29T17:16:57Z","snapshot_observed_at":"2026-08-01T16:30:31.998321Z","submitted_at":"2026-05-11T18:41:12Z","title":"Spurious Correlation Learning in Preference Optimization: Mechanisms, Consequences, and Mitigation via Tie Training","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-05-13T06:30:51.812541Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2605.11134"},"observation_digest":"sha256:abd56dcae879d808d7f941b3c3f2917de45a3f08ab5ab9cf498651b3d180f77f","observation_id":"59f00d15-ee8a-4cfe-8867-7658461bc1a0","resolution":{"observed_at":"2026-05-13T06:32:24.273715Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-09T15:30:16.931351Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":"2403.00409","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-07-04T11:09:46.406911Z","title":"Provably robust dpo: Aligning language models with noisy feed- back","venue":null,"work_id":"79a01a14-9a55-47f3-85eb-38feebf8582b","year":2024},"citing_paper":{"arxiv_id":"2605.23398","last_updated":"2026-05-22T09:11:20Z","snapshot_observed_at":"2026-08-02T13:58:02.707128Z","submitted_at":"2026-05-22T09:11:20Z","title":"TPMM-DPO: Trajectory-aware Preference-guided Model Merging for Iterative Direct Preference Optimization","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-25T03:41:52.859647Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2605.23398"},"observation_digest":"sha256:ffb08aea07e78d75c8cf5107d38853b09afd4c2a3f1970f093c1237f236dcc5a","observation_id":"a29d9f3b-da26-431f-b3c0-00d7387e32e4","resolution":{"observed_at":"2026-05-25T03:45:17.586731Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-09T15:30:16.931351Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":"2403.00409","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-07-04T11:09:46.406911Z","title":"Provably robust dpo: Aligning language models with noisy feed- back","venue":null,"work_id":"79a01a14-9a55-47f3-85eb-38feebf8582b","year":2024},"citing_paper":{"arxiv_id":"2606.19607","last_updated":"2026-06-17T21:19:01Z","snapshot_observed_at":"2026-08-02T22:14:26.237081Z","submitted_at":"2026-06-17T21:19:01Z","title":"Which Pairs to Compare for LLM Post-Training?","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-06-26T20:37:19.221848Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2606.19607"},"observation_digest":"sha256:0b69f1b954c358804fbe6b1e6c2d146f3b8a1dab8d9ff3c7e5168723e8a27bd8","observation_id":"66a041cc-61ca-4ecb-90bd-fc72154bdac8","resolution":{"observed_at":"2026-07-04T01:09:19.254368Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-09T15:30:16.931351Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":"2403.00409","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-07-04T11:09:46.406911Z","title":"Provably robust dpo: Aligning language models with noisy feed- back","venue":null,"work_id":"79a01a14-9a55-47f3-85eb-38feebf8582b","year":2024},"citing_paper":{"arxiv_id":"2606.24937","last_updated":"2026-07-27T15:17:17Z","snapshot_observed_at":"2026-08-02T23:19:25.465662Z","submitted_at":"2026-06-22T17:48:54Z","title":"The Hitchhiker's Guide to Agentic AI: From Foundations to Systems","version":1},"reference_index":190,"source":"pdf_text","source_observed_at":"2026-06-26T08:09:57.542558Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2606.24937"},"observation_digest":"sha256:2009894a6fb7bb55922fc2d739254409e71686a289935457a2e55fe1471cb549","observation_id":"0d736ac7-41f0-4a83-9703-22e57c8659ea","resolution":{"observed_at":"2026-07-04T11:09:46.408648Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-09T15:30:16.931351Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-08-02T10:27:18.410065Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback.arXiv Preprint arXiv:2403.00409, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.24937","last_updated":"2026-07-27T15:17:17Z","snapshot_observed_at":"2026-08-02T23:19:25.465662Z","submitted_at":"2026-06-22T17:48:54Z","title":"The Hitchhiker's Guide to Agentic AI: From Foundations to Systems","version":2},"reference_index":190,"source":"pdf_text","source_observed_at":"2026-08-02T10:27:18.410065Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2606.24937"},"observation_digest":"sha256:e84d17323f19c2795ea4a50043e814d822ee45ea37fc5c4b38f9e6414208985c","observation_id":"23266883-97f4-4d2f-8f72-348a70fd816d","resolution":{"observed_at":"2026-08-02T10:27:18.410065Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-09T15:30:16.931351Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-07-14T15:37:01.391649Z","title":"Provably robust dpo: Aligning language models with noisy feedback","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.09796","last_updated":"2026-07-20T03:41:16Z","snapshot_observed_at":"2026-08-09T19:08:22.057789Z","submitted_at":"2026-07-09T09:20:25Z","title":"Metadata-Free Meta-Reweighted Direct Preference Optimization under Noisy Preference Labels","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-14T15:37:01.391649Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2607.09796"},"observation_digest":"sha256:8c7e980948ffef20fbdb82cfc30c025dffe54b01271ffb43312d7d1cd7b6b0d4","observation_id":"f903dc6c-5203-4d51-8355-c19cb17ee593","resolution":{"observed_at":"2026-07-14T15:37:01.391649Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-09T15:30:16.931351Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-08-02T08:00:53.937134Z","title":"Provably robust dpo: Aligning language models with noisy feedback,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.09796","last_updated":"2026-07-20T03:41:16Z","snapshot_observed_at":"2026-08-09T19:08:22.057789Z","submitted_at":"2026-07-09T09:20:25Z","title":"Metadata-Free Meta-Reweighted Direct Preference Optimization under Noisy Preference Labels","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-02T08:00:53.937134Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2607.09796"},"observation_digest":"sha256:de1cc36872e36fe9c2bfe6f8c727ee596a5a2f3f5597608a37f4273b42be7579","observation_id":"1b17ee32-4086-4ec1-9545-b8151cf549fd","resolution":{"observed_at":"2026-08-02T08:00:53.937134Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-09T15:30:16.931351Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-08-08T15:03:47.464904Z","title":"arXiv preprint arXiv:2403.00409 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.05341","last_updated":"2026-08-05T18:59:23Z","snapshot_observed_at":"2026-08-09T23:10:36.410444Z","submitted_at":"2026-08-05T18:59:23Z","title":"Positive-Unlabeled Preference Optimization For Chest X-ray Report Generation","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-08T15:03:47.464904Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2608.05341"},"observation_digest":"sha256:eaa2dee385837363653128bf68409aa82962744899f2256bd1e6ef89c455a671","observation_id":"bff2af43-1386-4c93-8032-bc0d540df63a","resolution":{"observed_at":"2026-08-08T15:03:47.464904Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2403.00409/citation-record","integrity":"/paper/2403.00409/integrity","json":"/paper/2403.00409/citation-record.json","paper":"/paper/2403.00409"},"outbound":[],"paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-09T15:30:16.931351Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 21 inbound Pith citation observations for arXiv:2403.00409."}