{"as_of":"2026-08-20T05:08:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:741fbf9c815971c3f39f76a320b6395ea6d1bd8f778b4d27a7cb64b5210f467d","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":12,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":12,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":12,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":12,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T14:33:57.516039Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T12:06:55.872592Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2402.07314","last_updated":"2024-11-12T08:24:10Z","snapshot_observed_at":"2026-08-17T14:39:06.178537Z","submitted_at":"2024-02-11T21:44:21Z","title":"Online Iterative Reinforcement Learning from Human Feedback with General Preference Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07314","snapshot_observed_at":"2026-08-12T17:05:15.209576Z","title":", Xiong, W","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.13611","last_updated":"2024-12-10T07:47:15Z","snapshot_observed_at":"2026-08-16T17:58:55.737221Z","submitted_at":"2024-11-20T02:03:16Z","title":"DSTC: Direct Preference Learning with Only Self-Generated Tests and Code to Improve Code LMs","version":3},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-12T17:05:15.209576Z"},"links":{"cited_paper":"/paper/2402.07314","citing_paper":"/paper/2411.13611"},"observation_digest":"sha256:f897faf4ccaf6e0e9a8dba8879767d4367622b95eeab04250b2786fe81ee609b","observation_id":"1db72c5e-c4f9-4a0f-a0a4-ca77ce1a4faf","resolution":{"observed_at":"2026-08-12T17:05:15.209576Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07314","last_updated":"2024-11-12T08:24:10Z","snapshot_observed_at":"2026-08-17T14:39:06.178537Z","submitted_at":"2024-02-11T21:44:21Z","title":"Online Iterative Reinforcement Learning from Human Feedback with General Preference Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07314","snapshot_observed_at":"2026-08-12T00:11:14.827757Z","title":"A theoretical analysis of Nash learning from human feedback under general KL-regularized preference.arXiv:2402.07314,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.01951","last_updated":"2024-12-04T14:20:21Z","snapshot_observed_at":"2026-08-18T01:49:25.380130Z","submitted_at":"2024-12-02T20:24:17Z","title":"Self-Improvement in Language Models: The Sharpening Mechanism","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-12T00:11:14.827757Z"},"links":{"cited_paper":"/paper/2402.07314","citing_paper":"/paper/2412.01951"},"observation_digest":"sha256:e562afc36145c7903901afba57a589ccc412c47c00f9394eab697f44e26d2e67","observation_id":"e98b4f2e-9bb9-48cf-aa18-cc9c55bd8095","resolution":{"observed_at":"2026-08-12T00:11:14.827757Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07314","last_updated":"2024-11-12T08:24:10Z","snapshot_observed_at":"2026-08-17T14:39:06.178537Z","submitted_at":"2024-02-11T21:44:21Z","title":"Online Iterative Reinforcement Learning from Human Feedback with General Preference Model","version":3},"cited_work":{"arxiv_id":"2402.07314","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07314","snapshot_observed_at":"2026-07-02T12:06:55.872592Z","title":"Online iterative reinforce- ment learning from human feedback with general preference model","venue":null,"work_id":"edebf0a7-c19f-404d-9aa6-17412caa380e","year":2024},"citing_paper":{"arxiv_id":"2412.09413","last_updated":"2024-12-22T10:44:13Z","snapshot_observed_at":"2026-08-16T16:04:15.848709Z","submitted_at":"2024-12-12T16:20:36Z","title":"Imitate, Explore, and Self-Improve: A Reproduction Report on Slow-thinking Reasoning Systems","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-18T00:35:31.375020Z"},"links":{"cited_paper":"/paper/2402.07314","citing_paper":"/paper/2412.09413"},"observation_digest":"sha256:9452d8940b904e476c37d26d2ce46ec9fa6239e585828b9228fc828cc394c9d0","observation_id":"800496e3-91e4-41d9-88cc-260b1ff6e4e5","resolution":{"observed_at":"2026-05-18T00:35:31.494836Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07314","last_updated":"2024-11-12T08:24:10Z","snapshot_observed_at":"2026-08-17T14:39:06.178537Z","submitted_at":"2024-02-11T21:44:21Z","title":"Online Iterative Reinforcement Learning from Human Feedback with General Preference Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07314","snapshot_observed_at":"2026-08-11T15:55:20.419053Z","title":"A theoretical analysis of Nash learning from human feedback under general KL-regularized preference.arXiv preprint arXiv:2402.07314,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.10616","last_updated":"2024-12-13T23:42:24Z","snapshot_observed_at":"2026-08-19T14:39:35.664842Z","submitted_at":"2024-12-13T23:42:24Z","title":"Hybrid Preference Optimization for Alignment: Provably Faster Convergence Rates by Combining Offline Preferences with Online Exploration","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-11T15:55:20.419053Z"},"links":{"cited_paper":"/paper/2402.07314","citing_paper":"/paper/2412.10616"},"observation_digest":"sha256:81217fdfbba91b0a7121594341dd6ae2ca07499d7b19777af1ce3e105abad4f6","observation_id":"f56f94a3-d0d3-4a03-a3a8-816047500123","resolution":{"observed_at":"2026-08-11T15:55:20.419053Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07314","last_updated":"2024-11-12T08:24:10Z","snapshot_observed_at":"2026-08-17T14:39:06.178537Z","submitted_at":"2024-02-11T21:44:21Z","title":"Online Iterative Reinforcement Learning from Human Feedback with General Preference Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07314","snapshot_observed_at":"2026-08-07T15:15:46.376259Z","title":"A theoretical analysis of nash learning from human feedback under general kl-regularized preference","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17122","last_updated":"2025-05-21T17:59:02Z","snapshot_observed_at":"2026-08-18T01:28:05.129848Z","submitted_at":"2025-05-21T17:59:02Z","title":"Shallow Preference Signals: Large Language Model Aligns Even Better with Truncated Data?","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:46.376259Z"},"links":{"cited_paper":"/paper/2402.07314","citing_paper":"/paper/2505.17122"},"observation_digest":"sha256:92a770ec4e768cb552aab875ccf935eeb93686851964dc264bc100825367893f","observation_id":"4a78435b-0abb-422c-95b9-2722526d7a4a","resolution":{"observed_at":"2026-08-07T15:15:46.376259Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07314","last_updated":"2024-11-12T08:24:10Z","snapshot_observed_at":"2026-08-17T14:39:06.178537Z","submitted_at":"2024-02-11T21:44:21Z","title":"Online Iterative Reinforcement Learning from Human Feedback with General Preference Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07314","snapshot_observed_at":"2026-08-07T14:10:43.394782Z","title":"A theoretical analysis of nash learning from human feedback under general kl-regularized preference","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20268","last_updated":"2025-07-24T14:21:12Z","snapshot_observed_at":"2026-08-17T11:27:40.773999Z","submitted_at":"2025-05-26T17:44:08Z","title":"Outcome-Based Online Reinforcement Learning: Algorithms and Fundamental Limits","version":2},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-07T14:10:43.394782Z"},"links":{"cited_paper":"/paper/2402.07314","citing_paper":"/paper/2505.20268"},"observation_digest":"sha256:44dbddb940918308d3da67de7b74ec568ed2481d03e9b453e65c6e93d6d9aa2c","observation_id":"62417c3d-95fc-47b8-9e9b-f5ba558409e3","resolution":{"observed_at":"2026-08-07T14:10:43.394782Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07314","last_updated":"2024-11-12T08:24:10Z","snapshot_observed_at":"2026-08-17T14:39:06.178537Z","submitted_at":"2024-02-11T21:44:21Z","title":"Online Iterative Reinforcement Learning from Human Feedback with General Preference Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07314","snapshot_observed_at":"2026-08-06T19:28:59.337652Z","title":"A theoretical analysis of nash learning from human feedback under general kl-regularized preference","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.05913","last_updated":"2025-07-08T11:59:48Z","snapshot_observed_at":"2026-08-19T08:33:20.904937Z","submitted_at":"2025-07-08T11:59:48Z","title":"Best-of-N through the Smoothing Lens: KL Divergence and Regret Analysis","version":1},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-06T19:28:59.337652Z"},"links":{"cited_paper":"/paper/2402.07314","citing_paper":"/paper/2507.05913"},"observation_digest":"sha256:934cb69522cfa8909342979a795d45548ef6a0319bf6da1696874f4d68850c5c","observation_id":"683f5707-3ded-4ea2-9359-4f792525dc08","resolution":{"observed_at":"2026-08-06T19:28:59.337652Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07314","last_updated":"2024-11-12T08:24:10Z","snapshot_observed_at":"2026-08-17T14:39:06.178537Z","submitted_at":"2024-02-11T21:44:21Z","title":"Online Iterative Reinforcement Learning from Human Feedback with General Preference Model","version":3},"cited_work":{"arxiv_id":"2402.07314","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07314","snapshot_observed_at":"2026-07-02T12:06:55.872592Z","title":"Online iterative reinforce- ment learning from human feedback with general preference model","venue":null,"work_id":"edebf0a7-c19f-404d-9aa6-17412caa380e","year":2024},"citing_paper":{"arxiv_id":"2605.09214","last_updated":"2026-05-09T23:17:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-09T23:17:46Z","title":"Fast Rates for Offline Contextual Bandits with Forward-KL Regularization under Single-Policy Concentrability","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-05-12T03:47:14.379908Z"},"links":{"cited_paper":"/paper/2402.07314","citing_paper":"/paper/2605.09214"},"observation_digest":"sha256:d5bd879d597f89832a8cd306f09ecf5198c27eca4babcfb6d9ac396661d84617","observation_id":"681a11cf-4fc8-4558-aaed-db5ba133eae9","resolution":{"observed_at":"2026-05-12T06:56:31.027663Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07314","last_updated":"2024-11-12T08:24:10Z","snapshot_observed_at":"2026-08-17T14:39:06.178537Z","submitted_at":"2024-02-11T21:44:21Z","title":"Online Iterative Reinforcement Learning from Human Feedback with General Preference Model","version":3},"cited_work":{"arxiv_id":"2402.07314","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07314","snapshot_observed_at":"2026-07-02T12:06:55.872592Z","title":"Online iterative reinforce- ment learning from human feedback with general preference model","venue":null,"work_id":"edebf0a7-c19f-404d-9aa6-17412caa380e","year":2024},"citing_paper":{"arxiv_id":"2605.09363","last_updated":"2026-05-10T06:23:19Z","snapshot_observed_at":"2026-08-03T01:34:15.771125Z","submitted_at":"2026-05-10T06:23:19Z","title":"Near-Optimal Last-Iterate Convergence for Zero-Sum Games with Bandit Feedback and Opponent Actions","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-05-12T03:46:35.908522Z"},"links":{"cited_paper":"/paper/2402.07314","citing_paper":"/paper/2605.09363"},"observation_digest":"sha256:28774e3084c686d9da402f80c71902b515bacb9d84e6764499777752c4abd1df","observation_id":"15262199-a874-49e4-8ca7-17e4457fbee4","resolution":{"observed_at":"2026-05-12T06:56:31.615988Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07314","last_updated":"2024-11-12T08:24:10Z","snapshot_observed_at":"2026-08-17T14:39:06.178537Z","submitted_at":"2024-02-11T21:44:21Z","title":"Online Iterative Reinforcement Learning from Human Feedback with General Preference Model","version":3},"cited_work":{"arxiv_id":"2402.07314","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07314","snapshot_observed_at":"2026-07-02T12:06:55.872592Z","title":"Online iterative reinforce- ment learning from human feedback with general preference model","venue":null,"work_id":"edebf0a7-c19f-404d-9aa6-17412caa380e","year":2024},"citing_paper":{"arxiv_id":"2606.06053","last_updated":"2026-07-12T09:46:20Z","snapshot_observed_at":"2026-08-15T09:30:13.878235Z","submitted_at":"2026-06-04T11:54:23Z","title":"Online KL-Regularized Reinforcement Learning with Function Approximation under Misspecification","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-28T02:31:11.200818Z"},"links":{"cited_paper":"/paper/2402.07314","citing_paper":"/paper/2606.06053"},"observation_digest":"sha256:c31867bf543b54e15254b6a7eef2d78843ee32a3351c155c7b7872a9cd749e43","observation_id":"c02ab51e-a296-4c6f-ab53-b9a3347391f8","resolution":{"observed_at":"2026-07-02T12:06:55.873915Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07314","last_updated":"2024-11-12T08:24:10Z","snapshot_observed_at":"2026-08-17T14:39:06.178537Z","submitted_at":"2024-02-11T21:44:21Z","title":"Online Iterative Reinforcement Learning from Human Feedback with General Preference Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07314","snapshot_observed_at":"2026-07-14T18:23:21.126826Z","title":"Tong Zhang.Mathematical Analysis of Machine Learning Algorithms","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.06053","last_updated":"2026-07-12T09:46:20Z","snapshot_observed_at":"2026-08-15T09:30:13.878235Z","submitted_at":"2026-06-04T11:54:23Z","title":"Online KL-Regularized Reinforcement Learning with Function Approximation under Misspecification","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-07-14T18:23:21.126826Z"},"links":{"cited_paper":"/paper/2402.07314","citing_paper":"/paper/2606.06053"},"observation_digest":"sha256:12b815914cb22feac9b0ebde8290027475e92e4dffb78a735368793b878414ec","observation_id":"d4255a95-1077-4dd9-8b92-ceef7e6f0dd4","resolution":{"observed_at":"2026-07-14T18:23:21.126826Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07314","last_updated":"2024-11-12T08:24:10Z","snapshot_observed_at":"2026-08-17T14:39:06.178537Z","submitted_at":"2024-02-11T21:44:21Z","title":"Online Iterative Reinforcement Learning from Human Feedback with General Preference Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07314","snapshot_observed_at":"2026-08-15T14:33:57.516039Z","title":"arXiv preprint arXiv:2402.07314 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.07419","last_updated":"2026-08-07T17:05:10Z","snapshot_observed_at":"2026-08-19T11:38:58.290988Z","submitted_at":"2026-08-07T17:05:10Z","title":"Beyond Post-Hoc Temperature Scaling: Bilevel Optimization for LLM Calibration","version":1},"reference_index":161,"source":"arxiv_source","source_observed_at":"2026-08-15T14:33:57.516039Z"},"links":{"cited_paper":"/paper/2402.07314","citing_paper":"/paper/2608.07419"},"observation_digest":"sha256:38519b4fb2e2988ba7770570ca6640c7bceaecccc608fd94a98265c2c9b49c3b","observation_id":"02081b23-56b9-48b2-9542-1b49feb4a21a","resolution":{"observed_at":"2026-08-15T14:33:57.516039Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2402.07314/citation-record","integrity":"/paper/2402.07314/integrity","json":"/paper/2402.07314/citation-record.json","paper":"/paper/2402.07314"},"outbound":[],"paper":{"arxiv_id":"2402.07314","last_updated":"2024-11-12T08:24:10Z","latest_version":3,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-17T14:39:06.178537Z","submitted_at":"2024-02-11T21:44:21Z","title":"Online Iterative Reinforcement Learning from Human Feedback with General Preference Model"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 20 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 12 inbound Pith citation observations for arXiv:2402.07314."}