{"as_of":"2026-08-10T00:00:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:62dd94784dcf6f357e98138746c21c5df4931116ede4599100ecb25dab85b07d","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":39,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":39,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":39,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":39,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T20:34:01.215406Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":14,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2403.07691","last_updated":"2024-03-14T07:47:08Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-12T14:34:08Z","title":"ORPO: Monolithic Preference Optimization without Reference Model","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-05-16T09:34:04.394588Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2403.07691"},"observation_digest":"sha256:4f2f1a7f59e1c1d908dead50be4fd78d73dbe1946aa5ada0f8ecb97d775060c1","observation_id":"b045bfd9-05d4-4f25-8c19-72a780e85c46","resolution":{"observed_at":"2026-05-16T09:34:04.769240Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-09T20:34:01.215406Z","title":"G., Guo, Z","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.19358","last_updated":"2025-06-02T16:30:23Z","snapshot_observed_at":"2026-08-09T20:35:40.499093Z","submitted_at":"2025-01-31T18:10:53Z","title":"The Energy Loss Phenomenon in RLHF: A New Perspective on Mitigating Reward Hacking","version":3},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-09T20:34:01.215406Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2501.19358"},"observation_digest":"sha256:12d2803674ef61c0e5e59ad7e6b7766dde5960d89c3397faff942ace47eb5773","observation_id":"5aaa6f2a-ec60-4d7e-bc62-efde2d93a408","resolution":{"observed_at":"2026-08-09T20:34:01.215406Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-09T16:18:40.577727Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036, 2023","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.01208","last_updated":"2025-06-20T10:54:05Z","snapshot_observed_at":"2026-08-09T16:35:47.837184Z","submitted_at":"2025-02-03T09:59:32Z","title":"On Almost Surely Safe Alignment of Large Language Models at Inference-Time","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-09T16:18:40.577727Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2502.01208"},"observation_digest":"sha256:5e73e28f4a817ecbf4861a9c5d7812eb5a8ae2b731ed541d8d0db55144962ae1","observation_id":"7699aa7b-11f8-4f6e-8770-38fe3745ab72","resolution":{"observed_at":"2026-08-09T16:18:40.577727Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2502.01456","last_updated":"2025-09-26T09:25:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-03T15:43:48Z","title":"Process Reinforcement through Implicit Rewards","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-05-11T20:23:30.763794Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2502.01456"},"observation_digest":"sha256:a3c170dd2677230519096abae36e2c13c8bd788190cd041db5832d56558efe77","observation_id":"972de14a-91e6-49d4-94cb-6ccdaf5c574b","resolution":{"observed_at":"2026-05-11T20:23:30.944982Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-09T04:52:40.534383Z","title":"Bachmann, R., Kar, O","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.03429","last_updated":"2025-02-05T18:21:03Z","snapshot_observed_at":"2026-08-09T07:25:07.259449Z","submitted_at":"2025-02-05T18:21:03Z","title":"On Fairness of Unified Multimodal Large Language Model for Image Generation","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-09T04:52:40.534383Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2502.03429"},"observation_digest":"sha256:8c41ed8543df82dc96fe15c911fc48e784aff965a077b663ec69ebbe4a0b6f68","observation_id":"cf191c35-876d-4119-a643-6416b43f25a9","resolution":{"observed_at":"2026-08-09T04:52:40.534383Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-07T15:07:59.145850Z","title":"A general theoretical paradigm to understand learning from human preferences","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.17134","last_updated":"2025-06-03T03:04:17Z","snapshot_observed_at":"2026-08-09T07:25:09.275709Z","submitted_at":"2025-05-22T04:05:02Z","title":"LongMagpie: A Self-synthesis Method for Generating Large-scale Long-context Instructions","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-07T15:07:59.145850Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2505.17134"},"observation_digest":"sha256:d11eb50ac073d38e89f31d3f6ee705f038b8cd25274b4b58bccf0d56e5341372","observation_id":"b3153567-2340-453f-ad62-cc1375bc3e29","resolution":{"observed_at":"2026-08-07T15:07:59.145850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-07T14:31:45.997262Z","title":"A general theoretical paradigm to understand learning from human preferences","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.21537","last_updated":"2025-05-24T09:07:13Z","snapshot_observed_at":"2026-08-08T02:37:32.325481Z","submitted_at":"2025-05-24T09:07:13Z","title":"OpenReview Should be Protected and Leveraged as a Community Asset for Research in the Era of Large Language Models","version":1},"reference_index":129,"source":"arxiv_source","source_observed_at":"2026-08-07T14:31:45.997262Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2505.21537"},"observation_digest":"sha256:127e9053781150f55a2eb615709a59b6a4165000efe980869de322dd1e65192c","observation_id":"54a2c4d8-ace3-40d0-979d-4e89e722de36","resolution":{"observed_at":"2026-08-07T14:31:45.997262Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-07T12:43:47.923648Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.23927","last_updated":"2025-05-29T18:22:02Z","snapshot_observed_at":"2026-08-09T08:37:01.244014Z","submitted_at":"2025-05-29T18:22:02Z","title":"Thompson Sampling in Online RLHF with General Function Approximation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T12:43:47.923648Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2505.23927"},"observation_digest":"sha256:2b023b9a41b3f1f1771051cebae046325d353f4d896ef156d7f0a8bbb9502b18","observation_id":"f93d914f-0c4c-473a-82e8-2c013645e3ae","resolution":{"observed_at":"2026-08-07T12:43:47.923648Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-07T10:57:42.228327Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.03827","last_updated":"2025-06-04T10:57:18Z","snapshot_observed_at":"2026-08-08T01:31:46.984934Z","submitted_at":"2025-06-04T10:57:18Z","title":"Multi-objective Aligned Bidword Generation Model for E-commerce Search Advertising","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T10:57:42.228327Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2506.03827"},"observation_digest":"sha256:bf635f9b5ea48d890cb36f0f167a81d1232f768fb12a56fa263f9d1b1f5a0e72","observation_id":"c9abefb6-8aaf-4792-96b9-8cd2dca89181","resolution":{"observed_at":"2026-08-07T10:57:42.228327Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-07T05:22:07.333606Z","title":"Mohammad Gheshlaghi Azar, Mark Rowland, Bilal Piot, Daniel Guo, Daniele Calandriello, Michal Valko, and R´emi Munos","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.08379","last_updated":"2025-06-10T02:43:47Z","snapshot_observed_at":"2026-08-09T07:25:54.896697Z","submitted_at":"2025-06-10T02:43:47Z","title":"Reinforce LLM Reasoning through Multi-Agent Reflection","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-07T05:22:07.333606Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2506.08379"},"observation_digest":"sha256:4684c1ad4851b2416ff8ca2a29f75d28ef652673de66d9d077182749c84b23cf","observation_id":"38a7ec8d-1f7a-4786-babf-f21180eb9c63","resolution":{"observed_at":"2026-08-07T05:22:07.333606Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-07T00:16:06.358214Z","title":"Scaling direct preference optimization for fine-grained reward specification.arXiv preprint arXiv:2310.12036, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.14903","last_updated":"2025-06-17T18:17:35Z","snapshot_observed_at":"2026-08-09T07:25:11.092427Z","submitted_at":"2025-06-17T18:17:35Z","title":"DETONATE: A Benchmark for Text-to-Image Alignment and Kernelized Direct Preference Optimization","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T00:16:06.358214Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2506.14903"},"observation_digest":"sha256:559c9dc84849a6fc5f563c04eb0a668594594558714962fb7b1fdc4c078ccfaf","observation_id":"b2c59bc0-5896-4dd9-aef2-605091779e68","resolution":{"observed_at":"2026-08-07T00:16:06.358214Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-06T19:18:40.283348Z","title":"Scaling laws for reward model overoptimization.arXiv preprint arXiv:2310.12036, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.06013","last_updated":"2025-07-08T14:17:07Z","snapshot_observed_at":"2026-08-07T22:57:58.555368Z","submitted_at":"2025-07-08T14:17:07Z","title":"CogniSQL-R1-Zero: Lightweight Reinforced Reasoning for Efficient SQL Generation","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T19:18:40.283348Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2507.06013"},"observation_digest":"sha256:70dae1927b43f784ddf44a8d0d9436558ceeec1379140f3018cbb7c9f77857b9","observation_id":"fe1636fe-aa28-4d0a-a15d-841dcc051593","resolution":{"observed_at":"2026-08-06T19:18:40.283348Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2508.04149","last_updated":"2026-05-16T09:55:19Z","snapshot_observed_at":"2026-07-06T22:08:36.543090Z","submitted_at":"2025-08-06T07:24:14Z","title":"Difficulty-Based Preference Data Selection by DPO Implicit Reward Gap","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-21T23:46:24.208438Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2508.04149"},"observation_digest":"sha256:b045ff0342fb67f03e700c918eaf4e256bffa48ec141b7fae879aa49f0e77569","observation_id":"e1575096-4be4-46e7-b026-bd9c440d36d0","resolution":{"observed_at":"2026-05-21T23:50:47.607776Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T21:33:13.809159Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.08466","last_updated":"2025-08-11T20:53:37Z","snapshot_observed_at":"2026-08-09T07:25:49.374464Z","submitted_at":"2025-08-11T20:53:37Z","title":"Enhancing Small LLM Alignment through Margin-Based Objective Modifications under Resource Constraints","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-05T21:33:13.809159Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2508.08466"},"observation_digest":"sha256:8b6e6b5dbcd8b203b92deca4a039895c90ae64d0b42269ff78027e963a92ac44","observation_id":"fec1937c-8fd6-4e10-b526-f01cd2d18c30","resolution":{"observed_at":"2026-08-05T21:33:13.809159Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2509.20265","last_updated":"2026-04-29T12:32:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-09-24T15:52:36Z","title":"Failure Modes of Maximum Entropy RLHF","version":3},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-05-18T14:02:11.084514Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2509.20265"},"observation_digest":"sha256:cccfa80dc7f6fbbb3502658456a9c9f472e8aafbf2527bc33f301096fc2e8dfe","observation_id":"4e7e3696-975a-4dd5-9d5b-d36ed9ffd8d3","resolution":{"observed_at":"2026-05-18T14:02:39.813225Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-04T14:52:18.668425Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.22851","last_updated":"2026-07-02T18:42:52Z","snapshot_observed_at":"2026-08-09T07:25:07.650510Z","submitted_at":"2025-09-26T19:03:24Z","title":"Adaptive Margin RLHF via Preference over Preferences","version":4},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-04T14:52:18.668425Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2509.22851"},"observation_digest":"sha256:91545045c2dd4eab04a0ca83de86551cb672b3a125641b6ac44fd3c787e15017","observation_id":"4deb9cf5-bfb4-4a4e-a885-a13706c67041","resolution":{"observed_at":"2026-08-04T14:52:18.668425Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-03T14:24:19.502646Z","title":"A general theoretical paradigm to understand learning from human preferences, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2512.20806","last_updated":"2026-05-31T13:11:43Z","snapshot_observed_at":"2026-08-06T07:38:24.259000Z","submitted_at":"2025-12-23T22:13:14Z","title":"Safety Alignment of LMs via Non-cooperative Games","version":3},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-03T14:24:19.502646Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2512.20806"},"observation_digest":"sha256:cfc584dbf2cbef026f863ce97c9d2034d39d6827e668add432a1ef8aecb7f302","observation_id":"d058d0b3-5f42-4060-9235-c76e24e0b6db","resolution":{"observed_at":"2026-08-03T14:24:19.502646Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2605.02626","last_updated":"2026-05-04T14:15:24Z","snapshot_observed_at":"2026-07-30T07:44:04.614445Z","submitted_at":"2026-05-04T14:15:24Z","title":"Gradient-Gated DPO: Stabilizing Preference Optimization in Language Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-08T18:35:13.659698Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2605.02626"},"observation_digest":"sha256:bb896b8078f0ee7c8e54ad93ad325c6ef48e47bc900da18c5a32e7ee3e8b7380","observation_id":"03710aaf-d50c-4e56-b8a4-670bcc2e4707","resolution":{"observed_at":"2026-05-09T06:20:41.683804Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2605.06987","last_updated":"2026-05-07T22:05:23Z","snapshot_observed_at":"2026-07-31T05:30:22.249144Z","submitted_at":"2026-05-07T22:05:23Z","title":"Response Time Enhances Alignment with Heterogeneous Preferences","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-05-11T01:04:26.288913Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2605.06987"},"observation_digest":"sha256:b05362d9fefbfee6abd842cecd162a97f01c13992cc7d44bde8ae8b45d69506e","observation_id":"ae04bdcb-ef5f-4f03-9cd2-af5fee1a1754","resolution":{"observed_at":"2026-05-11T01:05:50.004199Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2605.11726","last_updated":"2026-05-13T15:38:02Z","snapshot_observed_at":"2026-07-06T23:23:30.499023Z","submitted_at":"2026-05-12T08:09:42Z","title":"Block-R1: Rethinking the Role of Block Size in Multi-domain Reinforcement Learning for Diffusion Large Language Models","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-13T07:03:00.503644Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2605.11726"},"observation_digest":"sha256:7b32660ebd49eccaa67c6680c6f30c5ee5d1f0df092ba503051b1920954338ad","observation_id":"2deb40d5-6e9a-486a-9ba4-13dc4827f5c1","resolution":{"observed_at":"2026-05-13T07:07:28.007161Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2605.11726","last_updated":"2026-05-13T15:38:02Z","snapshot_observed_at":"2026-07-06T23:23:30.499023Z","submitted_at":"2026-05-12T08:09:42Z","title":"Block-R1: Rethinking the Role of Block Size in Multi-domain Reinforcement Learning for Diffusion Large Language Models","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-14T21:06:01.667173Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2605.11726"},"observation_digest":"sha256:9961327dbca11453c8dc430fe89325b1232395bfe569bd6f0f0a6a99a2e80e89","observation_id":"36ad412e-a310-41da-b084-7a7d3f8d7d7d","resolution":{"observed_at":"2026-05-14T21:19:28.313009Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2605.11906","last_updated":"2026-05-12T10:18:49Z","snapshot_observed_at":"2026-07-31T22:11:05.515608Z","submitted_at":"2026-05-12T10:18:49Z","title":"YFPO: A Preliminary Study of Yoked Feature Preference Optimization with Neuron-Guided Rewards for Mathematical Reasoning","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-05-13T05:29:42.381280Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2605.11906"},"observation_digest":"sha256:1f9f21eb7f01d24b035b30c499141e5519b23de9e000d5edd960f964ddb48820","observation_id":"89c015fc-013f-4f68-85d5-8f127abb6416","resolution":{"observed_at":"2026-05-13T05:32:19.327417Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2605.12288","last_updated":"2026-06-10T07:32:21Z","snapshot_observed_at":"2026-07-06T23:23:59.123377Z","submitted_at":"2026-05-12T15:44:33Z","title":"TokenRatio: Principled Token-Level Preference Optimization via Ratio Matching","version":1},"reference_index":136,"source":"arxiv_source","source_observed_at":"2026-05-13T04:55:55.013900Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2605.12288"},"observation_digest":"sha256:0162d04b413e5113926ae5c33b2df1c1275639e3d46e595240a1ccf6d11c86b9","observation_id":"4b7e7949-a768-47e2-85a4-642bbdd8a483","resolution":{"observed_at":"2026-05-13T04:57:17.181807Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2605.12288","last_updated":"2026-06-10T07:32:21Z","snapshot_observed_at":"2026-07-06T23:23:59.123377Z","submitted_at":"2026-05-12T15:44:33Z","title":"TokenRatio: Principled Token-Level Preference Optimization via Ratio Matching","version":2},"reference_index":136,"source":"arxiv_source","source_observed_at":"2026-05-15T05:41:10.714594Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2605.12288"},"observation_digest":"sha256:0a194eff8e70769d686d06b6efa7fd4bd1961dcc817d8748700bdf23a0498d04","observation_id":"09a58498-7564-4110-a7c0-603eedba08a4","resolution":{"observed_at":"2026-05-15T05:45:06.641433Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2605.21854","last_updated":"2026-06-06T09:58:33Z","snapshot_observed_at":"2026-08-02T07:58:30.433073Z","submitted_at":"2026-05-21T01:02:41Z","title":"CrossVLA: Cross-Paradigm Post-Training and Inference Optimization for Vision-Language-Action Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-22T08:07:51.697353Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2605.21854"},"observation_digest":"sha256:e6abc8524b968a1773cc0a737e83138eb0552a19c2a06bfba6ba76b5bda86672","observation_id":"ce4112a6-da47-454b-a8cf-6dbba167bf8f","resolution":{"observed_at":"2026-05-22T08:11:17.214389Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2605.21854","last_updated":"2026-06-06T09:58:33Z","snapshot_observed_at":"2026-08-02T07:58:30.433073Z","submitted_at":"2026-05-21T01:02:41Z","title":"CrossVLA: Cross-Paradigm Post-Training and Inference Optimization for Vision-Language-Action Models","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-30T17:48:02.228941Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2605.21854"},"observation_digest":"sha256:12c341287c05c5df5caceb5c8a3e7b1786a91288a149a8cd5fcea70571dbd317","observation_id":"735f9d5d-4baf-4dcc-be85-c42485d78836","resolution":{"observed_at":"2026-07-01T15:05:47.998440Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2605.28440","last_updated":"2026-05-27T13:05:49Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T13:05:49Z","title":"AdaDPO: Self-Adaptive Direct Preference Optimization with Balanced Gradient Updates","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-29T12:29:55.729913Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2605.28440"},"observation_digest":"sha256:1926840e0b87b8562b2dd93e511998d3871f12040ecfac73d44a23d815891511","observation_id":"5b41e8e6-8bf3-4891-842a-8fd6f1a8a774","resolution":{"observed_at":"2026-06-29T12:33:24.402718Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2606.03089","last_updated":"2026-06-16T07:22:07Z","snapshot_observed_at":"2026-07-06T23:43:22.513369Z","submitted_at":"2026-06-02T03:17:56Z","title":"Constitutional On-Policy Safe Distillation","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-28T11:47:14.135793Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2606.03089"},"observation_digest":"sha256:c004b75dc0d244dab9c1bfe459e0f057f0b395647b27b5764df7d9b3997690e4","observation_id":"bb7add3f-651c-4f56-bc78-9cd92d032cec","resolution":{"observed_at":"2026-07-02T01:36:25.512570Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2606.05468","last_updated":"2026-06-03T21:47:43Z","snapshot_observed_at":"2026-07-06T23:45:28.379468Z","submitted_at":"2026-06-03T21:47:43Z","title":"FlowPRO: Reward-Free Reinforced Fine-Tuning of Flow-Matching VLAs via Proximalized Preference Optimization","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-28T05:38:11.089753Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2606.05468"},"observation_digest":"sha256:ba968b87565502a43c6ae2e888584605388366fded3e39634aa9ea372e4c6401","observation_id":"75c04942-f7a4-49e3-81b1-6bda7cfbccb3","resolution":{"observed_at":"2026-07-02T09:06:49.361855Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2606.23740","last_updated":"2026-06-21T15:34:21Z","snapshot_observed_at":"2026-08-02T05:47:42.942051Z","submitted_at":"2026-06-21T15:34:21Z","title":"Weight-Space Geometry of Offline Reasoning Training","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-06-26T10:26:28.702213Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2606.23740"},"observation_digest":"sha256:614ba037904f267ca219fb033172496a3b2ebed9b58109956c6d87cfdad4b1eb","observation_id":"5b049f08-6e51-467c-9c92-342a0f7eb1a2","resolution":{"observed_at":"2026-07-04T09:09:43.385341Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2606.24004","last_updated":"2026-06-29T07:01:12Z","snapshot_observed_at":"2026-08-02T11:15:29.427964Z","submitted_at":"2026-06-22T23:21:55Z","title":"Towards Spec Learning: Inference-Time Alignment from Preference Pairs","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-26T07:49:36.816100Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2606.24004"},"observation_digest":"sha256:a298f9aa3c759ef243db5cd7b9aa1d0ce74b64d15be90647db1a014f60df5200","observation_id":"a11da4f5-e823-4ed4-a425-56b5519e8320","resolution":{"observed_at":"2026-07-04T11:39:47.161717Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2606.24004","last_updated":"2026-06-29T07:01:12Z","snapshot_observed_at":"2026-08-02T11:15:29.427964Z","submitted_at":"2026-06-22T23:21:55Z","title":"Towards Spec Learning: Inference-Time Alignment from Preference Pairs","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-30T10:17:33.176525Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2606.24004"},"observation_digest":"sha256:819253bf8d782e919afe8265a1d7e2230afa4568069fe744bdff00a35775b8fa","observation_id":"38364724-f099-488a-a77c-71f5162f8b76","resolution":{"observed_at":"2026-06-30T12:44:39.556729Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2606.24937","last_updated":"2026-07-27T15:17:17Z","snapshot_observed_at":"2026-08-02T23:19:25.465662Z","submitted_at":"2026-06-22T17:48:54Z","title":"The Hitchhiker's Guide to Agentic AI: From Foundations to Systems","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-26T08:09:57.542558Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2606.24937"},"observation_digest":"sha256:016fc9db7bcc5cbf33a8a94bbd00f5bed63931aeea5f438f71290513425e5f80","observation_id":"a1437ddd-5690-4ed6-a8bc-fa94eab0e2ef","resolution":{"observed_at":"2026-07-04T11:09:46.374927Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-02T10:27:16.169555Z","title":"A General Theoretical Paradigm to Understand Learning from Human Feedback.arXiv Preprint arXiv:2310.12036, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.24937","last_updated":"2026-07-27T15:17:17Z","snapshot_observed_at":"2026-08-02T23:19:25.465662Z","submitted_at":"2026-06-22T17:48:54Z","title":"The Hitchhiker's Guide to Agentic AI: From Foundations to Systems","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-02T10:27:16.169555Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2606.24937"},"observation_digest":"sha256:c631b68c9f87490ca635654a694e86d7345cced391fad831c3dd8f8937584e1b","observation_id":"b6fc0ef8-98a3-4acf-9eb1-8386c9e100f9","resolution":{"observed_at":"2026-08-02T10:27:16.169555Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-07-12T04:43:45.592808Z","title":"arXiv preprint arXiv:2310.12036 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.03126","last_updated":"2026-07-07T03:50:51Z","snapshot_observed_at":"2026-08-02T17:17:56.686166Z","submitted_at":"2026-07-03T09:14:27Z","title":"ACPO: Adaptive Credit Policy Optimization via Fine-Grained Surrogate Entropy","version":2},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-07-12T04:43:45.592808Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2607.03126"},"observation_digest":"sha256:e46d0d8b53448f69538393e475bcac67c380486d65be9bf97caf79c9205b7ab7","observation_id":"a5e8bc7a-5eba-4e50-a3b7-20421d5bdd65","resolution":{"observed_at":"2026-07-12T04:43:45.592808Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-07-11T13:53:36.775836Z","title":"arXiv preprint arXiv:2310.12036 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.04763","last_updated":"2026-07-26T14:17:18Z","snapshot_observed_at":"2026-08-02T10:24:43.977557Z","submitted_at":"2026-07-06T07:56:53Z","title":"Multi-Turn On-Policy Distillation with Prefix Replay","version":1},"reference_index":163,"source":"arxiv_source","source_observed_at":"2026-07-11T13:53:36.775836Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2607.04763"},"observation_digest":"sha256:8442fe722049dd3af4cdf8b18b6755c097f4f68ef12771c054b9b4104be13304","observation_id":"a8fe755a-582d-4e88-b4ca-db7b229fe51b","resolution":{"observed_at":"2026-07-11T13:53:36.775836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-02T08:40:51.116483Z","title":"arXiv preprint arXiv:2310.12036 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.04763","last_updated":"2026-07-26T14:17:18Z","snapshot_observed_at":"2026-08-02T10:24:43.977557Z","submitted_at":"2026-07-06T07:56:53Z","title":"Multi-Turn On-Policy Distillation with Prefix Replay","version":3},"reference_index":164,"source":"arxiv_source","source_observed_at":"2026-08-02T08:40:51.116483Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2607.04763"},"observation_digest":"sha256:ae588c1a24f2176bb0e1788732b2700c03b81c52b6c9c05fbb10881dcae6cfa8","observation_id":"b839b03e-a437-4a0d-a884-49efdc099734","resolution":{"observed_at":"2026-08-02T08:40:51.116483Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-07-31T12:20:07.021635Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.28282","last_updated":"2026-07-30T14:31:07Z","snapshot_observed_at":"2026-08-06T16:34:19.905250Z","submitted_at":"2026-07-30T14:31:07Z","title":"(Towards) Scalable Reliable Automated Evaluation with Large Language Models","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-07-31T12:20:07.021635Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2607.28282"},"observation_digest":"sha256:201394cc985596da780e8a0b6797ee1aaf81292f563230e144ca6b8eb0d9b9f8","observation_id":"8899f132-d89f-4e05-bac8-94441a09aef5","resolution":{"observed_at":"2026-07-31T12:20:07.021635Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-07T00:13:50.414984Z","title":"A general theoretical paradigm to understand learning from human preferences","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.02713","last_updated":"2026-08-03T17:59:58Z","snapshot_observed_at":"2026-08-07T23:09:56.055766Z","submitted_at":"2026-08-03T17:59:58Z","title":"Quo Vadis, World Modeling?","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T00:13:50.414984Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2608.02713"},"observation_digest":"sha256:6ff1d24e0f2605871daee4ac38c4ac0b2e641d8f3625b2459ee2ad6678385da2","observation_id":"a51d4882-beae-4871-b73f-73d1a086b0f0","resolution":{"observed_at":"2026-08-07T00:13:50.414984Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2310.12036/citation-record","integrity":"/paper/2310.12036/integrity","json":"/paper/2310.12036/citation-record.json","paper":"/paper/2310.12036"},"outbound":[],"paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","latest_version":2,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-09T07:24:36.881438Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 39 inbound Pith citation observations for arXiv:2310.12036."}