{"as_of":"2026-08-15T09:06:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:0685f7c1dde2ec4084c0266530580745cb709bd497fb2d2ed398f42d68513b9f","coverage":[{"denominator":31,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":31,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T01:04:33.188878Z","state":"measured"},{"denominator":34,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":34,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T22:50:33.214346Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T06:49:37.712152Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.12350","snapshot_observed_at":"2026-08-07T22:50:33.214346Z","title":"Theoretical tensions in rlhf: Reconciling empirical success with inconsistencies in social choice theory","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2502.09053","last_updated":"2025-08-05T02:23:31Z","snapshot_observed_at":"2026-08-08T10:12:47.586452Z","submitted_at":"2025-02-13T08:08:27Z","title":"Game Theory Meets Large Language Models: A Systematic Survey with Taxonomy and New Frontiers","version":2},"reference_index":151,"source":"pdf_text","source_observed_at":"2026-08-07T22:50:33.214346Z"},"links":{"cited_paper":"/paper/2506.12350","citing_paper":"/paper/2502.09053"},"observation_digest":"sha256:596e8e0ce1d2a2b18b7c07f163ef2eac0ea427c8dff3cc6a735f97a88c6143ed","observation_id":"5deb5269-7ec9-4442-b42c-f9b9117dcbea","resolution":{"observed_at":"2026-08-07T22:50:33.214346Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"cited_work":{"arxiv_id":"2506.12350","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.12350","snapshot_observed_at":"2026-07-04T06:49:37.712152Z","title":null,"venue":null,"work_id":"e6f1a23e-33e3-431b-a8ab-e51d7cdd6f75","year":2025},"citing_paper":{"arxiv_id":"2606.02340","last_updated":"2026-06-01T14:48:19Z","snapshot_observed_at":"2026-08-07T21:01:14.231398Z","submitted_at":"2026-06-01T14:48:19Z","title":"Transitivity in Inhomogeneous Random Tournaments","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-06-28T12:45:56.736538Z"},"links":{"cited_paper":"/paper/2506.12350","citing_paper":"/paper/2606.02340"},"observation_digest":"sha256:d4b9046122a5a3d3e68590167cded3bea7b63fbd63d1fb31c1180ac0c881ddff","observation_id":"381d7141-7902-4833-9a51-9c632f59e13b","resolution":{"observed_at":"2026-07-02T01:06:24.252873Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"cited_work":{"arxiv_id":"2506.12350","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.12350","snapshot_observed_at":"2026-07-04T06:49:37.712152Z","title":null,"venue":null,"work_id":"e6f1a23e-33e3-431b-a8ab-e51d7cdd6f75","year":2025},"citing_paper":{"arxiv_id":"2606.21550","last_updated":"2026-06-19T15:47:01Z","snapshot_observed_at":"2026-08-14T09:57:59.632970Z","submitted_at":"2026-06-19T15:47:01Z","title":"AI Alignment From Social Choice Perspectives","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-06-26T14:12:36.892697Z"},"links":{"cited_paper":"/paper/2506.12350","citing_paper":"/paper/2606.21550"},"observation_digest":"sha256:369dc1c8c1ed14bf6e5f7911c4814ad06572613f49206fac0a70f8f1b0c469f3","observation_id":"c78663c3-2f78-44a3-b449-3e9eaf0d3262","resolution":{"observed_at":"2026-07-04T06:49:37.714149Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.12350/citation-record","integrity":"/paper/2506.12350/integrity","json":"/paper/2506.12350/citation-record.json","paper":"/paper/2506.12350"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:04:34.710047Z","title":"Therefore, a preference matching distribution exists","venue":null,"work_id":"3ffd90a4-3e50-4372-a9d9-0bef2f47381b","year":2023},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:32.970029Z"},"links":{"citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:255698f6b4cd7a04e081abdfcc4b6b08c5c19193d3a0d3082afba12754ac06cc","observation_id":"5b11d9d1-86ef-42c9-8247-4415029ad08c","resolution":{"observed_at":"2026-08-07T01:04:34.815076Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.10271","last_updated":"2024-06-04T14:34:38Z","snapshot_observed_at":"2026-08-15T06:46:24.373292Z","submitted_at":"2024-04-16T03:59:33Z","title":"Social Choice Should Guide AI Alignment in Dealing with Diverse Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.10271","snapshot_observed_at":"2026-08-07T01:04:30.630057Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:30.630057Z"},"links":{"cited_paper":"/paper/2404.10271","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:266207024c8268aa1a2c9f86b8ee441b24dc27fdf1d8261ad6fc2e88782a2efd","observation_id":"d6efb15f-2b56-4a38-b874-2a29f44bad6b","resolution":{"observed_at":"2026-08-07T01:04:30.630057Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.13038","last_updated":"2024-04-19T17:49:56Z","snapshot_observed_at":"2026-08-13T00:25:52.838460Z","submitted_at":"2024-04-19T17:49:56Z","title":"Mapping Social Choice Theory to RLHF","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.13038","snapshot_observed_at":"2026-08-07T01:04:30.717742Z","title":"Mapping social choice theory to RLHF.arXiv preprint arXiv:2404.13038,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:30.717742Z"},"links":{"cited_paper":"/paper/2404.13038","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:06b5383455330de8b36f7b018a5f4ad8c5c35b81263c1f41a2ed117dce22e374","observation_id":"a701d99e-19f1-4c20-b602-d115cabc1f85","resolution":{"observed_at":"2026-08-07T01:04:30.717742Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.10584","last_updated":"2024-02-25T19:15:26Z","snapshot_observed_at":"2026-08-13T05:00:13.831621Z","submitted_at":"2023-12-17T02:14:15Z","title":"Policy Optimization in RLHF: The Impact of Out-of-preference Data","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.10584","snapshot_observed_at":"2026-08-07T01:04:31.048119Z","title":"Policy optimization in rlhf: The impact of out-of-preference data","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:31.048119Z"},"links":{"cited_paper":"/paper/2312.10584","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:469e62b92993bea05556c50317db2eeaa869e1e049968f33ed9acd15ed320705","observation_id":"37cfa10b-53f2-4d10-807e-fffd6a485e01","resolution":{"observed_at":"2026-08-07T01:04:31.048119Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.19266","last_updated":"2025-01-31T16:26:28Z","snapshot_observed_at":"2026-08-14T23:17:31.796138Z","submitted_at":"2025-01-31T16:26:28Z","title":"Jackpot! Alignment as a Maximal Lottery","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.19266","snapshot_observed_at":"2026-08-07T01:04:31.269934Z","title":"Jackpot! alignment as a maximal lottery.arXiv preprint arXiv:2501.19266,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:31.269934Z"},"links":{"cited_paper":"/paper/2501.19266","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:af89d3e67d41e8f722da7e1dbe08e896f85e5e6477e5f215784a3b0523b1e4c9","observation_id":"e1530569-83dd-4e7b-9891-ac9968612789","resolution":{"observed_at":"2026-08-07T01:04:31.269934Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16048","last_updated":"2023-10-24T17:59:04Z","snapshot_observed_at":"2026-08-13T05:41:56.448068Z","submitted_at":"2023-10-24T17:59:04Z","title":"AI Alignment and Social Choice: Fundamental Limitations and Policy Implications","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.16048","snapshot_observed_at":"2026-08-07T01:04:31.412637Z","title":"AI alignment and social choice: Fundamental limitations and policy implications","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:31.412637Z"},"links":{"cited_paper":"/paper/2310.16048","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:9b4a836308f95e530ab31a416e770f78df7df91c6173d2a6bfb16ac398717dde","observation_id":"f056459f-c02d-4a49-91fb-13dadd438e7c","resolution":{"observed_at":"2026-08-07T01:04:31.412637Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.00886","last_updated":"2024-06-11T16:25:52Z","snapshot_observed_at":"2026-08-13T05:12:12.105399Z","submitted_at":"2023-12-01T19:26:23Z","title":"Nash Learning from Human Feedback","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.00886","snapshot_observed_at":"2026-08-07T01:04:31.531656Z","title":"Nash learning from human feedback.arXiv preprint arXiv:2312.00886,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:31.531656Z"},"links":{"cited_paper":"/paper/2312.00886","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:5502e2d37d26a323b5a50ca220aa6f5aa3172699db4eff78dd162fde07d56ce8","observation_id":"af2407b8-2d5e-41c8-a244-5ddb9e26359a","resolution":{"observed_at":"2026-08-07T01:04:31.531656Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00254","last_updated":"2024-05-27T14:08:40Z","snapshot_observed_at":"2026-08-13T00:17:57.625735Z","submitted_at":"2024-04-30T23:57:23Z","title":"RLHF from Heterogeneous Feedback via Personalization and Preference Aggregation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00254","snapshot_observed_at":"2026-08-07T01:04:31.735488Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:31.735488Z"},"links":{"cited_paper":"/paper/2405.00254","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:89e2a538a658bfb4f1ebc3d09f531dce4a44f7357316fefc3c825d6b4400be0f","observation_id":"88259934-527d-46dc-b40b-0a6c150812d2","resolution":{"observed_at":"2026-08-07T01:04:31.735488Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-07T01:04:31.830315Z","title":"John Schulman, Filip Wolski, Prafulla Dhariwal, Alec Radford, and Oleg Klimov","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:31.830315Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:b57ac70ccb686fb882d0c9cb4543d29f7c16541f8ed10c5c2e7a9f46fe9bd30f","observation_id":"d60b0dca-4b57-46e9-82f0-746fb3ec9b80","resolution":{"observed_at":"2026-08-07T01:04:31.830315Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.08358","last_updated":"2024-04-17T01:58:09Z","snapshot_observed_at":"2026-08-13T05:03:03.153876Z","submitted_at":"2023-12-13T18:51:34Z","title":"Distributional Preference Learning: Understanding and Accounting for Hidden Context in RLHF","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.08358","snapshot_observed_at":"2026-08-07T01:04:32.028901Z","title":"Distributional preference learn- ing: Understanding and accounting for hidden context in RLHF.arXiv preprint arXiv:2312.08358,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:32.028901Z"},"links":{"cited_paper":"/paper/2312.08358","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:31a063aca5d756219ebf17f31e0c4b24d05ef679588ce747c9c8ce362fca7d22","observation_id":"699e2fc7-0744-4fe0-9bf4-189127a6bc72","resolution":{"observed_at":"2026-08-07T01:04:32.028901Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.08448","last_updated":"2024-05-14T09:12:30Z","snapshot_observed_at":"2026-08-13T00:55:26.938356Z","submitted_at":"2024-05-14T09:12:30Z","title":"Understanding the performance gap between online and offline alignment algorithms","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.08448","snapshot_observed_at":"2026-08-07T01:04:32.093625Z","title":"Understanding the perfor- mance gap between online and offline alignment algorithms.arXiv preprint arXiv:2405.08448,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:32.093625Z"},"links":{"cited_paper":"/paper/2405.08448","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:109ab56a73537090dc28c048115417a138cd79124d04aab76339555688008427","observation_id":"92b86db4-ffbe-4594-be5f-5f1bfa313a7d","resolution":{"observed_at":"2026-08-07T01:04:32.093625Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-07T01:04:32.208926Z","title":"Gemini: a family of highly capable multimodal models.arXiv preprint arXiv:2312.11805,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:32.208926Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:adcd45810f8a7a0a749fa30ec2f40931ca876e96832c1ac1bffd12d4e262015d","observation_id":"f2767bef-1860-470f-b3ee-a9578e512d76","resolution":{"observed_at":"2026-08-07T01:04:32.208926Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-07T01:04:32.286520Z","title":"Llama: Open and efficient foundation language models.arXiv preprint arXiv:2302.13971,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:32.286520Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:901e843be6a1d48f103f76b7c7855d6dd40ca53222a682edad6607035fb93327","observation_id":"e894c17b-8779-4d88-b5c3-7ae66709f369","resolution":{"observed_at":"2026-08-07T01:04:32.286520Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:04:35.460658Z","title":"Zhilin Wang, Yi Dong, Jiaqi Zeng, Virginia Adams, Makesh Narsimhan Sreedhar, Daniel Egert, Olivier Delalleau, Jane Scowcroft, Neel Kant, Aidan Swope, et al","venue":null,"work_id":"0921af18-79d1-4a8f-bc7b-480d66ed1fb1","year":2024},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:32.350165Z"},"links":{"citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:2b951ec977f376a5f3acf596a90517dafdcdbbb76da8f09eb42ffb02ee7bfbd6","observation_id":"c8b38568-b696-4339-b24e-92ad0dda7007","resolution":{"observed_at":"2026-08-07T01:04:35.593726Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.16455","last_updated":"2025-08-25T01:01:38Z","snapshot_observed_at":"2026-08-12T23:57:53.587697Z","submitted_at":"2024-05-26T07:00:05Z","title":"On the Algorithmic Bias of Aligning Large Language Models with RLHF: Preference Collapse and Matching Regularization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.16455","snapshot_observed_at":"2026-08-07T01:04:32.454213Z","title":"On the algorithmic bias of aligning large language models with rlhf: Preference collapse and matching regularization.arXiv preprint arXiv:2405.16455,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:32.454213Z"},"links":{"cited_paper":"/paper/2405.16455","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:67c65e52a552e5e9b298a545eccf6170b71ac41da02da14b71507e1a91988f19","observation_id":"5f3c4032-9f7c-458e-964b-d3b0f333e6ba","resolution":{"observed_at":"2026-08-07T01:04:32.454213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:04:32.522528Z","title":"Restoring calibration for aligned large language models: A calibration-aware fine-tuning approach","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:32.522528Z"},"links":{"citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:df4739e3516b262be380113876f835eef27901039af23bb3bd17fccccf1a5e23","observation_id":"7a2cbe65-6812-4d08-8b31-c3678f0730a1","resolution":{"observed_at":"2026-08-07T01:04:32.522528Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:04:35.287569Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl- constraint","venue":null,"work_id":"b1af7fe7-1996-490c-b64b-1a07293b26e7","year":2024},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:32.636118Z"},"links":{"citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:40e3d293a6fd23e1f07c003b54b31582b0048de065401d08a17a0c2ca87c6460","observation_id":"7defe16f-f701-4713-bf0f-45a0706a7933","resolution":{"observed_at":"2026-08-07T01:04:35.373688Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:04:35.113189Z","title":"Asymptotics of language model alignment","venue":null,"work_id":"32c3c79d-9fd3-4947-9f23-4e138b965e06","year":2027},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:32.741501Z"},"links":{"citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:13be4b22e297063a7a029747cbde26a86088eeb4deffed337e740df844867cd6","observation_id":"b33ced5a-b656-45c1-8310-f2c9ba8bd1ee","resolution":{"observed_at":"2026-08-07T01:04:35.202878Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05006","last_updated":"2024-03-08T03:05:11Z","snapshot_observed_at":"2026-08-13T01:31:56.074944Z","submitted_at":"2024-03-08T03:05:11Z","title":"Provable Multi-Party Reinforcement Learning with Diverse Human Feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05006","snapshot_observed_at":"2026-08-07T01:04:32.828481Z","title":"Provable multi-party reinforcement learning with diverse human feedback.arXiv preprint arXiv:2403.05006,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:32.828481Z"},"links":{"cited_paper":"/paper/2403.05006","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:5cb401fbbc93ccd19e35f7c0e01c08bd114839a5abe6ebe3b61eb4ec9b0dbc82","observation_id":"3fcb66fe-0fc2-42c9-a6cf-fd8b5072d6a7","resolution":{"observed_at":"2026-08-07T01:04:32.828481Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:04:34.904498Z","title":null,"venue":null,"work_id":"4a957b9e-0011-4f1d-868f-c1120fe9688d","year":2024},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:32.907603Z"},"links":{"citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:46c98b64fddb5a4c6d106ca6ae9207881f4ff24229a09c5ed60e2ae4eace27a8","observation_id":"c87aacea-1141-4d61-b508-6f6cba18e980","resolution":{"observed_at":"2026-08-07T01:04:35.004266Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:04:34.504628Z","title":"The work by Tang et al","venue":null,"work_id":"27f8739e-91b8-426d-aca0-6976ab5065d1","year":2017},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:33.030370Z"},"links":{"citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:59ca38fb69cc8e77d1a1fc646b3a5625ad2ef42d34025180367adc0d31914e93","observation_id":"6ff25476-08d1-443b-8c00-679588b028d9","resolution":{"observed_at":"2026-08-07T01:04:34.587683Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:04:34.250344Z","title":"Outside of fine-tuning, model editing has emerged as a complementary strategy to modify LLM behavior across tasks [Jin et al., 2025]","venue":null,"work_id":"7470cf3e-ba0d-4db5-9b56-d2e700c84d52","year":2025},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:33.117798Z"},"links":{"citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:1311da9b83ef8fc2ec9cd0f0418e1d18bb21461c76b33d911d88b17a33405cda","observation_id":"861998dd-ec03-4cda-bb5b-b6c69066a382","resolution":{"observed_at":"2026-08-07T01:04:34.387321Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:04:34.078966Z","title":"[2024], which constrain its robustness relative to reinforcement learning techniques like PPO","venue":null,"work_id":"12c5a699-5442-47d6-9b87-325f6a2bf491","year":2024},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:33.188878Z"},"links":{"citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:02cba9f55a22a3c45242283c63f9ea94bcd3ffaf010f7a6d5c88fc13764c7ed1","observation_id":"7aa67290-0996-43c2-8634-c003f6d7a217","resolution":{"observed_at":"2026-08-07T01:04:34.151963Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09656","last_updated":"2025-02-25T10:19:35Z","snapshot_observed_at":"2026-08-13T14:21:51.626111Z","submitted_at":"2024-04-15T10:44:31Z","title":"Learn Your Reference Model for Real Good Alignment","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09656","snapshot_observed_at":"2026-08-07T01:04:30.926721Z","title":"Learn your reference model for real good alignment.arXiv preprint arXiv:2404.09656,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":2006,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:30.926721Z"},"links":{"cited_paper":"/paper/2404.09656","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:9902ae4f3025235f04c82ea7e71bf0e1ea9ba7d25e264202c42529e4f8469ef2","observation_id":"cad5854e-c103-4d4f-8d73-ebc7127bc9aa","resolution":{"observed_at":"2026-08-07T01:04:30.926721Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.02612","last_updated":"2024-06-19T17:08:13Z","snapshot_observed_at":"2026-08-14T03:09:05.589792Z","submitted_at":"2024-05-04T08:43:45Z","title":"Learning Linear Utility Functions From Pairwise Comparison Queries","version":3},"cited_work":{"arxiv_id":"2405.02612","doi":null,"metadata_source":"pith","pith_arxiv_id":"2405.02612","snapshot_observed_at":"2026-08-07T01:04:33.833571Z","title":"Learning Linear Utility Functions From Pairwise Comparison Queries","venue":"cs.LG","work_id":"a2bcb8f1-deb1-4c52-b209-82ffca95ca0c","year":2024},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":2009,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:30.810972Z"},"links":{"cited_paper":"/paper/2405.02612","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:dffc1f630511e4e44678bfeaae93dde95ae658d7e5ff5bfff59ed130dd3ba4bf","observation_id":"1b4f020e-eabd-446e-9266-f8a868b6dc13","resolution":{"observed_at":"2026-08-07T01:04:33.883031Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.12712","last_updated":"2023-04-13T20:41:31Z","snapshot_observed_at":"2026-08-03T04:49:15.195814Z","submitted_at":"2023-03-22T16:51:28Z","title":"Sparks of Artificial General Intelligence: Early experiments with GPT-4","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.12712","snapshot_observed_at":"2026-08-07T01:04:30.242850Z","title":"Sparks of artificial general intelligence: Early experiments with gpt-4.arXiv preprint arXiv:2303.12712,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":2016,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:30.242850Z"},"links":{"cited_paper":"/paper/2303.12712","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:ed8c7a6e470396e13e2ae9bbecf19a7668d2e4ad7ed0d58e4dda4687419de3ab","observation_id":"80dbad57-95b5-4a4b-a3b3-b85a5bc8976e","resolution":{"observed_at":"2026-08-07T01:04:30.242850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.20627","last_updated":"2025-05-27T02:07:35Z","snapshot_observed_at":"2026-08-11T02:01:04.493797Z","submitted_at":"2025-05-27T02:07:35Z","title":"Fundamental Limits of Game-Theoretic LLM Alignment: Smith Consistency and Preference Matching","version":1},"cited_work":{"arxiv_id":"2505.20627","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.20627","snapshot_observed_at":"2026-08-07T01:04:33.548624Z","title":"Fundamental Limits of Game-Theoretic LLM Alignment: Smith Consistency and Preference Matching","venue":"cs.GT","work_id":"2aac02d0-c8b6-4d01-9b77-d4f27398331c","year":2025},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:31.958892Z"},"links":{"cited_paper":"/paper/2505.20627","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:fded9039a7cbb1f31f05b2072f53c19009cc6b084d8b8ddd0cc597b969ac67fb","observation_id":"5f332d57-5d8f-4179-ba1e-cbc868b75180","resolution":{"observed_at":"2026-08-07T01:04:33.572321Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-07T01:04:31.655495Z","title":"Gpt-4 technical report.arXiv preprint arXiv:2303.08774,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:31.655495Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:9699d9a9ae8c18d8b7214b4998cee5960b0ef1538e3a8c61bda93ff6d560d085","observation_id":"0093612e-68fb-4bcc-999e-531b37b21540","resolution":{"observed_at":"2026-08-07T01:04:31.655495Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.08925","last_updated":"2024-12-26T00:15:20Z","snapshot_observed_at":"2026-08-15T03:25:32.357214Z","submitted_at":"2024-02-14T03:56:27Z","title":"MaxMin-RLHF: Alignment with Diverse Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.08925","snapshot_observed_at":"2026-08-07T01:04:30.349050Z","title":"URLhttps://openreview.net/forum?id=bx24KpJ4Eb","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:30.349050Z"},"links":{"cited_paper":"/paper/2402.08925","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:11f2ad8c8274dff167177cd2bb38fba312b67f08951fdbedf6050246b3dd78d1","observation_id":"074dd878-1664-4c9d-88dc-fa80b3d653da","resolution":{"observed_at":"2026-08-07T01:04:30.349050Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08495","last_updated":"2024-04-16T17:36:39Z","snapshot_observed_at":"2026-08-13T15:18:03.060033Z","submitted_at":"2024-04-12T14:25:49Z","title":"Dataset Reset Policy Optimization for RLHF","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.08495","snapshot_observed_at":"2026-08-07T01:04:30.492218Z","title":"Dataset reset policy optimization for rlhf.arXiv preprint arXiv:2404.08495,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:30.492218Z"},"links":{"cited_paper":"/paper/2404.08495","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:1a14568341c41c97f2c3b91ad2c9221eff70b54ee72f78ecffedd31c4cf1fa1c","observation_id":"19322493-ba98-49d0-bc1a-954f3af1adf5","resolution":{"observed_at":"2026-08-07T01:04:30.492218Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10990","last_updated":"2026-04-30T19:06:42Z","snapshot_observed_at":"2026-07-06T20:52:27.848221Z","submitted_at":"2025-03-14T01:29:21Z","title":"Statistical Impossibility and Possibility of Aligning LLMs with Human Preferences: From Condorcet Paradox to Nash Equilibrium","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10990","snapshot_observed_at":"2026-08-07T01:04:31.156730Z","title":"Kaizhao Liu, Qi Long, Zhekun Shi, Weijie J Su, and Jiancong Xiao","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:31.156730Z"},"links":{"cited_paper":"/paper/2503.10990","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:9a022fa75623bc98e65e51267344b39d25a2569d1c9dc835bc081d0597687661","observation_id":"8a24506e-5f96-4b25-8720-621acdde4d0f","resolution":{"observed_at":"2026-08-07T01:04:31.156730Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","latest_version":1,"primary_category":"stat.ML","snapshot_observed_at":"2026-08-08T15:24:03.755091Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory"},"reference_resolution":{"displayed":31,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":22,"verified_exact":1,"verified_fuzzy":7},"total_outbound_references":31},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 31 of 31 outbound references and 3 inbound Pith citation observations for arXiv:2506.12350."}