{"as_of":"2026-08-10T02:43:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e14aaaa79cd505458c863b5c49c2e60770ccbd72e1af3b42550bdc8023c6c591","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":33,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":33,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":33,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":33,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T11:47:17.645871Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T08:07:45.294768Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":"2312.11456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-07-03T08:07:45.294768Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456","venue":null,"work_id":"a7265aa9-5ce8-4a96-acfd-3f069003f21d","year":2024},"citing_paper":{"arxiv_id":"2401.10020","last_updated":"2025-03-28T00:06:51Z","snapshot_observed_at":"2026-08-07T08:02:34.857823Z","submitted_at":"2024-01-18T14:43:47Z","title":"Self-Rewarding Language Models","version":3},"reference_index":120,"source":"arxiv_source","source_observed_at":"2026-05-13T12:01:42.290502Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2401.10020"},"observation_digest":"sha256:317d4e32756470137a07eafddd55a935a1d311bd44df1f4260085e221fa00894","observation_id":"be79fbfa-d555-494a-8c4a-1fae5e1094ea","resolution":{"observed_at":"2026-05-13T12:01:42.513253Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-09T11:47:17.645871Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.04354","last_updated":"2025-02-04T18:47:11Z","snapshot_observed_at":"2026-08-09T19:08:38.829460Z","submitted_at":"2025-02-04T18:47:11Z","title":"Reviving The Classics: Active Reward Modeling in Large Language Model Alignment","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-09T11:47:17.645871Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2502.04354"},"observation_digest":"sha256:fc958da18b2878c1a42664b8ef837e51e1d0a8e79b0458c4c926fedf92fe16c4","observation_id":"68877fec-7230-405e-b79d-0453e484ba40","resolution":{"observed_at":"2026-08-09T11:47:17.645871Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-09T11:32:47.964012Z","title":"Gibbs sampling from human feedback: A prov- able kl-constrained framework for rlhf","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.04357","last_updated":"2025-02-04T19:37:35Z","snapshot_observed_at":"2026-08-09T21:31:34.898991Z","submitted_at":"2025-02-04T19:37:35Z","title":"Reusing Embeddings: Reproducible Reward Model Research in Large Language Model Alignment without GPUs","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-09T11:32:47.964012Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2502.04357"},"observation_digest":"sha256:667f6726ee8547553611aadcc3e466438908ce827a8b2771ce1e90674795967a","observation_id":"5d3c12fb-96dc-4564-94f4-9bdabdee110e","resolution":{"observed_at":"2026-08-09T11:32:47.964012Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-07T14:37:50.431221Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.18433","last_updated":"2025-08-13T01:12:42Z","snapshot_observed_at":"2026-08-09T06:26:17.359728Z","submitted_at":"2025-05-24T00:00:43Z","title":"Finite-Time Global Optimality Convergence in Deep Neural Actor-Critic Methods for Decentralized Multi-Agent Reinforcement Learning","version":2},"reference_index":4344,"source":"pdf_text","source_observed_at":"2026-08-07T14:37:50.431221Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2505.18433"},"observation_digest":"sha256:8629e69ea48fcf44c8dc34c8c797a95848086e529992b72c9222fa9a2d40cddc","observation_id":"cb628c24-4853-460c-a3d4-056fa4cffffd","resolution":{"observed_at":"2026-08-07T14:37:50.431221Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-07T14:01:06.783481Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.20556","last_updated":"2025-05-26T22:34:42Z","snapshot_observed_at":"2026-08-09T01:22:43.161220Z","submitted_at":"2025-05-26T22:34:42Z","title":"Learning a Pessimistic Reward Model in RLHF","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T14:01:06.783481Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2505.20556"},"observation_digest":"sha256:dbc1edba99e5f7dbe68c6de27437ec076018041d7d19d586b992e53d9371f55a","observation_id":"c925f773-402e-4005-aad0-6c67f4ba00c6","resolution":{"observed_at":"2026-08-07T14:01:06.783481Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-07T14:31:46.467622Z","title":"Gibbs sampling from human feedback: A provable kl-constrained framework for rlhf","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.21537","last_updated":"2025-05-24T09:07:13Z","snapshot_observed_at":"2026-08-08T02:37:32.325481Z","submitted_at":"2025-05-24T09:07:13Z","title":"OpenReview Should be Protected and Leveraged as a Community Asset for Research in the Era of Large Language Models","version":1},"reference_index":133,"source":"arxiv_source","source_observed_at":"2026-08-07T14:31:46.467622Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2505.21537"},"observation_digest":"sha256:5a4c8016f299533f01c88ec80d8d8fd1b071ca6ca9dfef4b6de198e0b3ab590c","observation_id":"e26b3256-2698-4ae6-8ffa-1f03b099f100","resolution":{"observed_at":"2026-08-07T14:31:46.467622Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-07T13:24:30.488898Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.21907","last_updated":"2025-05-31T04:48:02Z","snapshot_observed_at":"2026-08-09T21:30:35.237559Z","submitted_at":"2025-05-28T02:52:39Z","title":"Modeling and Optimizing User Preferences in AI Copilots: A Comprehensive Survey and Taxonomy","version":2},"reference_index":102,"source":"pdf_text","source_observed_at":"2026-08-07T13:24:30.488898Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2505.21907"},"observation_digest":"sha256:813545cbf9f3d9ddd71597b81b66cbffb217a7f3bfa0c83da5c824f3304004d1","observation_id":"2fd6201e-da41-4d52-bed5-5e89a29b156c","resolution":{"observed_at":"2026-08-07T13:24:30.488898Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-07T12:43:49.435213Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint.arXiv e-prints, page arXiv:2312.11456, December 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23927","last_updated":"2025-05-29T18:22:02Z","snapshot_observed_at":"2026-08-09T08:37:01.244014Z","submitted_at":"2025-05-29T18:22:02Z","title":"Thompson Sampling in Online RLHF with General Function Approximation","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T12:43:49.435213Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2505.23927"},"observation_digest":"sha256:a7867a6dc9455974ae6bad7ec9e4cc39c989c8422b9af813c11e025c84779a80","observation_id":"d633d1f9-2b01-4aa8-929c-e316d7ca4f71","resolution":{"observed_at":"2026-08-07T12:43:49.435213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-07T10:50:52.162823Z","title":"InForty-first International Conference on Machine Learning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.04463","last_updated":"2025-06-04T21:29:11Z","snapshot_observed_at":"2026-08-08T10:10:08.232794Z","submitted_at":"2025-06-04T21:29:11Z","title":"Aligning Large Language Models with Implicit Preferences from User-Generated Content","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T10:50:52.162823Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2506.04463"},"observation_digest":"sha256:18552f8a01dc83a1bb778b55a27419a2c359064a03696a8f89066e9b8fcd585f","observation_id":"738674b3-804c-466f-8cb8-2176cf94dc03","resolution":{"observed_at":"2026-08-07T10:50:52.162823Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-07T05:51:30.736946Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.06923","last_updated":"2025-06-07T21:23:00Z","snapshot_observed_at":"2026-08-07T12:11:21.205285Z","submitted_at":"2025-06-07T21:23:00Z","title":"Boosting LLM Reasoning via Spontaneous Self-Correction","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T05:51:30.736946Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2506.06923"},"observation_digest":"sha256:b92d509ba85cfc838ce8d6317157e40218f03946e62b081beecde0ffdfc71abf","observation_id":"2cbcdcb0-37db-4fba-b52b-61e85f8acf78","resolution":{"observed_at":"2026-08-07T05:51:30.736946Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-06T22:28:09.397544Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.21495","last_updated":"2025-06-26T17:25:49Z","snapshot_observed_at":"2026-08-06T22:21:21.361920Z","submitted_at":"2025-06-26T17:25:49Z","title":"Bridging Offline and Online Reinforcement Learning for LLMs","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-06T22:28:09.397544Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2506.21495"},"observation_digest":"sha256:821de703da81dd838bdfef3de3fed0f43c11d40df120f0b389253b61d2f06908","observation_id":"aadb1a06-6ba7-44f1-8ad0-d2b140535341","resolution":{"observed_at":"2026-08-06T22:28:09.397544Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":"2312.11456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-07-03T08:07:45.294768Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456","venue":null,"work_id":"a7265aa9-5ce8-4a96-acfd-3f069003f21d","year":2024},"citing_paper":{"arxiv_id":"2507.06419","last_updated":"2026-06-04T20:44:16Z","snapshot_observed_at":"2026-08-06T19:02:44.982053Z","submitted_at":"2025-07-08T21:56:33Z","title":"Teach a Reward Model to Correct Itself: Reward Guided Adversarial Failure Discovery for Robust Reward Modeling","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-05-19T05:16:22.274580Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2507.06419"},"observation_digest":"sha256:704b5222e91fa1bd1b3c4aa1f0ab7a01a0205c0ac26dc5a7bca843e45ffce6df","observation_id":"7a5f7fda-8506-4f53-83f0-03e83d0c13f5","resolution":{"observed_at":"2026-05-19T05:17:05.794667Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-06T16:34:25.249776Z","title":"Gibbs sampling from human feedback: A provable kl-constrained framework for rlhf.arXiv preprint arXiv:2312.11456,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.13158","last_updated":"2025-07-17T14:22:24Z","snapshot_observed_at":"2026-08-09T20:19:53.961381Z","submitted_at":"2025-07-17T14:22:24Z","title":"Inverse Reinforcement Learning Meets Large Language Model Post-Training: Basics, Advances, and Opportunities","version":1},"reference_index":101,"source":"pdf_text","source_observed_at":"2026-08-06T16:34:25.249776Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2507.13158"},"observation_digest":"sha256:8895990b1719343d402055863ac22e3ff78e1a35dc5b5834a1f42a403cf43f3e","observation_id":"caa7a167-a786-40b8-ad11-8e33e6c466dd","resolution":{"observed_at":"2026-08-06T16:34:25.249776Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":"2312.11456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-07-03T08:07:45.294768Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456","venue":null,"work_id":"a7265aa9-5ce8-4a96-acfd-3f069003f21d","year":2024},"citing_paper":{"arxiv_id":"2508.16771","last_updated":"2026-07-02T23:23:00Z","snapshot_observed_at":"2026-08-05T17:05:43.834611Z","submitted_at":"2025-08-22T20:08:09Z","title":"EyeMulator: Improving Code Language Models by Mimicking Human Visual Attention","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-18T20:54:30.449792Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2508.16771"},"observation_digest":"sha256:44f6fc7ed99332c61688f60e0fa9e90837811116fddc471d57a4731c25d17a48","observation_id":"fce8968b-b7ec-4828-97f6-67d3d1edf231","resolution":{"observed_at":"2026-05-18T20:56:51.807191Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":"2312.11456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-07-03T08:07:45.294768Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456","venue":null,"work_id":"a7265aa9-5ce8-4a96-acfd-3f069003f21d","year":2024},"citing_paper":{"arxiv_id":"2508.20697","last_updated":"2026-08-04T09:31:07Z","snapshot_observed_at":"2026-08-07T23:09:05.524726Z","submitted_at":"2025-08-28T12:07:11Z","title":"Token Buncher: Shielding LLMs from Harmful Reinforcement Learning Fine-Tuning","version":3},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-18T20:40:44.496392Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2508.20697"},"observation_digest":"sha256:8fdae7f4b6a144bee1083f891cef48ee4bd8798b01f42ab513b70f82f93665c2","observation_id":"e9f47699-f286-41a3-83c0-869043ed15e1","resolution":{"observed_at":"2026-05-18T20:41:50.556600Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-04T22:59:14.577593Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.06941","last_updated":"2025-09-08T17:52:56Z","snapshot_observed_at":"2026-08-07T12:11:53.268632Z","submitted_at":"2025-09-08T17:52:56Z","title":"Outcome-based Exploration for LLM Reasoning","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-04T22:59:14.577593Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2509.06941"},"observation_digest":"sha256:67a4eea17f4d9c2b9fa05bb304d13c2917c1fe4b2b3bf2ade809e6cfb5c1fe42","observation_id":"ca620d9e-6f09-4fc7-bfbe-5a69903cbd8f","resolution":{"observed_at":"2026-08-04T22:59:14.577593Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-04T16:07:44.508870Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.16679","last_updated":"2025-09-20T13:11:28Z","snapshot_observed_at":"2026-08-04T16:07:24.699834Z","submitted_at":"2025-09-20T13:11:28Z","title":"Reinforcement Learning Meets Large Language Models: A Survey of Advancements and Applications Across the LLM Lifecycle","version":1},"reference_index":207,"source":"pdf_text","source_observed_at":"2026-08-04T16:07:44.508870Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2509.16679"},"observation_digest":"sha256:b416f8f3e1ebb875eb0e7d9da7b22c7bd079db647ba06df8fa51a717c4ad6e1b","observation_id":"f46296f8-9f42-4f61-bc8a-4a54339c294d","resolution":{"observed_at":"2026-08-04T16:07:44.508870Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-04T14:57:00.278197Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.22992","last_updated":"2026-08-03T08:38:00Z","snapshot_observed_at":"2026-08-10T00:59:10.624197Z","submitted_at":"2025-09-26T23:08:03Z","title":"T-TAMER: Provably Taming Trade-offs in ML Serving","version":2},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-08-04T14:57:00.278197Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2509.22992"},"observation_digest":"sha256:0cb12b19e0fd9dbef00e30fcb150ec763a0e04815a7480ac25ca8ceb76678a27","observation_id":"54e2a028-9ce8-437c-82c9-b871889ae3d1","resolution":{"observed_at":"2026-08-04T14:57:00.278197Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":"2312.11456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-07-03T08:07:45.294768Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456","venue":null,"work_id":"a7265aa9-5ce8-4a96-acfd-3f069003f21d","year":2024},"citing_paper":{"arxiv_id":"2509.23102","last_updated":"2026-04-06T19:14:12Z","snapshot_observed_at":"2026-08-03T06:33:00.420621Z","submitted_at":"2025-09-27T04:18:33Z","title":"Multiplayer Nash Preference Optimization","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-18T13:09:54.433720Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2509.23102"},"observation_digest":"sha256:57dc6b84e27a79696c958a226909d93156c668b79fe2bc89a2d4bbdd2407c8ce","observation_id":"a187e3c4-ed2a-4fd9-bc9c-25ad2c792f07","resolution":{"observed_at":"2026-05-18T13:11:24.027237Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-03T13:42:06.121305Z","title":"Private reinforcement learning with pac and regret guarantees","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2512.23816","last_updated":"2026-06-25T06:56:14Z","snapshot_observed_at":"2026-08-06T05:05:37.952259Z","submitted_at":"2025-12-29T19:20:35Z","title":"Improved Bounds for Private and Robust Alignment","version":2},"reference_index":2000,"source":"pdf_text","source_observed_at":"2026-08-03T13:42:06.121305Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2512.23816"},"observation_digest":"sha256:fde157ca1a84bc353f713ad6232af5488b48f209c31050bc3bda53255f818fcb","observation_id":"29f80c5d-e9d3-4b69-a6ca-ff32d85aa70e","resolution":{"observed_at":"2026-08-03T13:42:06.121305Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":"2312.11456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-07-03T08:07:45.294768Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456","venue":null,"work_id":"a7265aa9-5ce8-4a96-acfd-3f069003f21d","year":2024},"citing_paper":{"arxiv_id":"2602.07832","last_updated":"2026-07-03T17:35:46Z","snapshot_observed_at":"2026-08-03T03:33:41.469348Z","submitted_at":"2026-02-08T05:47:27Z","title":"rePIRL: Learn PRM with Inverse RL for LLM Reasoning","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-21T13:13:13.293921Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2602.07832"},"observation_digest":"sha256:ec4a42bd0226f91fb4d0e0887c4caad8b8f1c7d45b342b13883a027ba7124afb","observation_id":"9f384787-e9db-4a2a-bf85-b581734215fd","resolution":{"observed_at":"2026-05-21T13:14:11.017458Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-03T03:33:45.088960Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.07832","last_updated":"2026-07-03T17:35:46Z","snapshot_observed_at":"2026-08-03T03:33:41.469348Z","submitted_at":"2026-02-08T05:47:27Z","title":"rePIRL: Learn PRM with Inverse RL for LLM Reasoning","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-03T03:33:45.088960Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2602.07832"},"observation_digest":"sha256:38861e5d61a232ab42fb2e34bcf3e97e024024c127468d1993e6b1b49dd387c0","observation_id":"51151045-dabe-41db-85a2-582cede2776b","resolution":{"observed_at":"2026-08-03T03:33:45.088960Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":"2312.11456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-07-03T08:07:45.294768Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456","venue":null,"work_id":"a7265aa9-5ce8-4a96-acfd-3f069003f21d","year":2024},"citing_paper":{"arxiv_id":"2604.13598","last_updated":"2026-04-15T08:08:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-15T08:08:06Z","title":"Enhancing Reinforcement Learning for Radiology Report Generation with Evidence-aware Rewards and Self-correcting Preference Learning","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-05-10T14:09:06.139625Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2604.13598"},"observation_digest":"sha256:2899baf5f521f7e1355e38887ce17a7518b3eeaaf76b6bf4d71c180686d93d73","observation_id":"11128f59-79a4-4dc7-90d6-57e1d37c1fe3","resolution":{"observed_at":"2026-05-10T14:10:28.664074Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":"2312.11456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-07-03T08:07:45.294768Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456","venue":null,"work_id":"a7265aa9-5ce8-4a96-acfd-3f069003f21d","year":2024},"citing_paper":{"arxiv_id":"2604.20933","last_updated":"2026-04-22T11:52:21Z","snapshot_observed_at":"2026-07-30T07:55:19.034385Z","submitted_at":"2026-04-22T11:52:21Z","title":"IRIS: Interpolative R\\'enyi Iterative Self-play for Large Language Model Fine-Tuning","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-05-10T01:15:12.985803Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2604.20933"},"observation_digest":"sha256:c73e55069af8790ff1088e9ca083bfe348e795670dfcc7ba35ac899b8711899e","observation_id":"40f61a43-74ac-4b71-8338-b529d1583785","resolution":{"observed_at":"2026-05-11T13:41:05.005307Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":"2312.11456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-07-03T08:07:45.294768Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456","venue":null,"work_id":"a7265aa9-5ce8-4a96-acfd-3f069003f21d","year":2024},"citing_paper":{"arxiv_id":"2605.01831","last_updated":"2026-05-03T11:45:08Z","snapshot_observed_at":"2026-07-06T23:15:01.968194Z","submitted_at":"2026-05-03T11:45:08Z","title":"RMGAP: Benchmarking the Generalization of Reward Models across Diverse Preferences","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-09T17:27:05.533755Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2605.01831"},"observation_digest":"sha256:1b9393753e91f64b1d9e07f262e09ed6fe5d14458c828da3616f4a2d0b1dd876","observation_id":"110138e7-d015-4b0d-a398-f9f63980401c","resolution":{"observed_at":"2026-05-11T16:21:07.087022Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":"2312.11456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-07-03T08:07:45.294768Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456","venue":null,"work_id":"a7265aa9-5ce8-4a96-acfd-3f069003f21d","year":2024},"citing_paper":{"arxiv_id":"2605.06977","last_updated":"2026-05-07T21:48:26Z","snapshot_observed_at":"2026-07-06T23:19:25.538514Z","submitted_at":"2026-05-07T21:48:26Z","title":"$f$-Divergence Regularized RLHF: Two Tales of Sampling and Unified Analyses","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-11T01:13:29.292351Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2605.06977"},"observation_digest":"sha256:45e7a45ccccc8fcd181f3b47a7e10d441660c2d33a6967b48666473a4d1f16f7","observation_id":"3ef6e794-1ad0-4748-a798-b34efe831837","resolution":{"observed_at":"2026-05-11T04:35:58.665306Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":"2312.11456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-07-03T08:07:45.294768Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456","venue":null,"work_id":"a7265aa9-5ce8-4a96-acfd-3f069003f21d","year":2024},"citing_paper":{"arxiv_id":"2605.11134","last_updated":"2026-05-29T17:16:57Z","snapshot_observed_at":"2026-08-01T16:30:31.998321Z","submitted_at":"2026-05-11T18:41:12Z","title":"Spurious Correlation Learning in Preference Optimization: Mechanisms, Consequences, and Mitigation via Tie Training","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-05-13T06:30:51.812541Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2605.11134"},"observation_digest":"sha256:3acbabee3b8b2be8c1476ee1a574686bebc6861cf797d5d799c937752d58f7b3","observation_id":"4ba08c5f-0a38-41ec-8355-2b6b0715ab24","resolution":{"observed_at":"2026-05-13T06:32:24.270894Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":"2312.11456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-07-03T08:07:45.294768Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456","venue":null,"work_id":"a7265aa9-5ce8-4a96-acfd-3f069003f21d","year":2024},"citing_paper":{"arxiv_id":"2606.03131","last_updated":"2026-06-02T04:18:08Z","snapshot_observed_at":"2026-08-02T20:05:30.693121Z","submitted_at":"2026-06-02T04:18:08Z","title":"HARVE: Hacking-Aware Reward-Head Vector Editing for Robust Reward Models","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-28T11:32:16.166724Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2606.03131"},"observation_digest":"sha256:db1a108dc8b47ebfb408f8e4be7c8333c7cd1bfaca76985b8278342d66dbd811","observation_id":"eb3dd61e-1eb5-4ff0-841d-65edf791a5a6","resolution":{"observed_at":"2026-07-02T01:46:26.783878Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":"2312.11456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-07-03T08:07:45.294768Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456","venue":null,"work_id":"a7265aa9-5ce8-4a96-acfd-3f069003f21d","year":2024},"citing_paper":{"arxiv_id":"2606.11437","last_updated":"2026-06-09T20:48:48Z","snapshot_observed_at":"2026-08-04T03:39:21.234561Z","submitted_at":"2026-06-09T20:48:48Z","title":"The Power of Test-Time Training for Approximate Sampling","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-27T11:07:47.543592Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2606.11437"},"observation_digest":"sha256:8246078cc102e31b0d4a741133fa40a1593e87621802b3bd1b6a4ed898057142","observation_id":"07b58e96-e25d-472f-80e9-06479c7bbd52","resolution":{"observed_at":"2026-07-03T08:07:45.296622Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":"2312.11456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-07-03T08:07:45.294768Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456","venue":null,"work_id":"a7265aa9-5ce8-4a96-acfd-3f069003f21d","year":2024},"citing_paper":{"arxiv_id":"2606.30445","last_updated":"2026-06-29T15:17:42Z","snapshot_observed_at":"2026-08-07T23:00:10.618237Z","submitted_at":"2026-06-29T15:17:42Z","title":"When Does Online Imitation Learning Help in LLM Post-Training? The Role of (Non-)Realizability Beyond Horizon","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-06-30T07:41:39.266071Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2606.30445"},"observation_digest":"sha256:cc5e472e682262b6059e25bae7b847544c2601ea3904e4a416b9771d4de281d0","observation_id":"cec4c83d-8ba5-45b8-b569-d375ceb5c8f1","resolution":{"observed_at":"2026-06-30T07:44:21.719763Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-01T23:59:41.739998Z","title":"mlr.press/v216/wimmer23a.html","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15196","last_updated":"2026-08-04T09:02:09Z","snapshot_observed_at":"2026-08-08T13:15:41.457205Z","submitted_at":"2026-07-16T16:52:54Z","title":"Subjective Risk Decomposition: A New View for Uncertainty Quantification","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-01T23:59:41.739998Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2607.15196"},"observation_digest":"sha256:b1527a5411f7f82cd6e7fd87f359e0fd05a6241c46e29edcef1b580bfc641f31","observation_id":"05b50ba8-3813-405f-8765-7a6df5cff1c2","resolution":{"observed_at":"2026-08-01T23:59:41.739998Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-02T10:01:59.330889Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.16240","last_updated":"2026-06-26T03:03:15Z","snapshot_observed_at":"2026-08-08T01:20:34.619049Z","submitted_at":"2026-06-26T03:03:15Z","title":"Normalized Rewards for Preference Optimization","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-02T10:01:59.330889Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2607.16240"},"observation_digest":"sha256:041df224219784ab15115e57d3c3b66db42d6da541de675b3aaa30d7a65b06a2","observation_id":"ab20eedc-4fd4-4669-abb6-94f61bdc20e0","resolution":{"observed_at":"2026-08-02T10:01:59.330889Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-08T15:03:47.550403Z","title":"arXiv preprint arXiv:2312.11456 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.05341","last_updated":"2026-08-05T18:59:23Z","snapshot_observed_at":"2026-08-09T23:10:36.410444Z","submitted_at":"2026-08-05T18:59:23Z","title":"Positive-Unlabeled Preference Optimization For Chest X-ray Report Generation","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-08T15:03:47.550403Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2608.05341"},"observation_digest":"sha256:75ed9ffa301b1d142d76760f45b971f4156aa69d2f0c3d8e3e1ca4bc5d05b4f2","observation_id":"801be598-9771-4bfb-a8e4-51aa7ce27033","resolution":{"observed_at":"2026-08-08T15:03:47.550403Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2312.11456/citation-record","integrity":"/paper/2312.11456/integrity","json":"/paper/2312.11456/citation-record.json","paper":"/paper/2312.11456"},"outbound":[],"paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","latest_version":4,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 33 inbound Pith citation observations for arXiv:2312.11456."}