{"as_of":"2026-08-23T04:13:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:fbe074d86beb3eb90cdf4314b52e91bd2f54f7fde4c526a1724c0202e1e890ac","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":19,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":19,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-22T06:32:14.747728+00:00","state":"measured"},{"denominator":19,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":19,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T12:31:57.941537Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-21T07:59:50.977785Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-08-17T08:58:04.443078Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-08-12T17:15:47.929415Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.12843","last_updated":"2024-11-19T20:17:04Z","snapshot_observed_at":"2026-08-17T01:00:48.700288Z","submitted_at":"2024-11-19T20:17:04Z","title":"Reward Modeling with Ordinal Feedback: Wisdom of the Crowd","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-12T17:15:47.929415Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2411.12843"},"observation_digest":"sha256:4feb62b174fd6aea3014ee6ca5485f04e861e5b859aa091e88e0faed30a1136b","observation_id":"0c1c3830-a08e-4dee-aea8-068e398d2bfe","resolution":{"observed_at":"2026-08-12T17:15:47.929415Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-08-17T08:58:04.443078Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-08-11T12:38:43.118721Z","title":"Arithmetic Control of LLMs for Diverse User Preferences : Directional Preference Alignment with Multi - Objective Rewards , 2024 a","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13998","last_updated":"2024-12-18T16:14:59Z","snapshot_observed_at":"2026-08-12T06:32:44.431104Z","submitted_at":"2024-12-18T16:14:59Z","title":"Few-shot Steerable Alignment: Adapting Rewards and LLM Policies with Neural Processes","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-11T12:38:43.118721Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2412.13998"},"observation_digest":"sha256:b759734ef0fbed5bfd79587ba722f48545263897449f9cec6ce4c40bf78f9bf1","observation_id":"da1bb913-e86b-4ec3-a51c-8e5b0c49a762","resolution":{"observed_at":"2026-08-11T12:38:43.118721Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-08-17T08:58:04.443078Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-08-09T11:47:17.638227Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.04354","last_updated":"2025-02-04T18:47:11Z","snapshot_observed_at":"2026-08-16T01:43:22.380380Z","submitted_at":"2025-02-04T18:47:11Z","title":"Reviving The Classics: Active Reward Modeling in Large Language Model Alignment","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-09T11:47:17.638227Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2502.04354"},"observation_digest":"sha256:9e1b287ac4cfa86d5baf792948c4b69e4aea3220ed351e56963641c24f808cc4","observation_id":"f269a302-d70b-4d9f-af17-0cb8f115d568","resolution":{"observed_at":"2026-08-09T11:47:17.638227Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-08-17T08:58:04.443078Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-08-08T11:20:48.489366Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.07956","last_updated":"2025-02-11T21:09:24Z","snapshot_observed_at":"2026-08-20T04:39:09.418555Z","submitted_at":"2025-02-11T21:09:24Z","title":"Bridging HCI and AI Research for the Evaluation of Conversational SE Assistants","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-08T11:20:48.489366Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2502.07956"},"observation_digest":"sha256:0f5877fe7950f2311f596f934ef472a8ec0b15e0be67816054c03bfc5b6019ee","observation_id":"b5c40558-960e-416e-b402-753a034c8944","resolution":{"observed_at":"2026-08-08T11:20:48.489366Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-08-17T08:58:04.443078Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-08-16T12:31:57.941537Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.12663","last_updated":"2025-06-11T07:31:37Z","snapshot_observed_at":"2026-08-17T13:10:48.080808Z","submitted_at":"2025-04-17T05:50:13Z","title":"Persona-judge: Personalized Alignment of Large Language Models via Token-level Self-judgment","version":2},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-16T12:31:57.941537Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2504.12663"},"observation_digest":"sha256:be914bfe81005fffe0993ac01f6669efaf0196e79568b57baf8423ae17b49697","observation_id":"8c8b266c-63c4-4646-98d7-6e4e80215d26","resolution":{"observed_at":"2026-08-16T12:31:57.941537Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-08-17T08:58:04.443078Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-08-15T21:08:56.923655Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.10892","last_updated":"2026-06-05T07:57:09Z","snapshot_observed_at":"2026-08-20T10:55:10.469429Z","submitted_at":"2025-05-16T05:58:26Z","title":"Multi-Objective Preference Optimization: Improving Human Alignment of Generative Models","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-15T21:08:56.923655Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2505.10892"},"observation_digest":"sha256:a5242cfb0ba3b503dc58dfe07457a6b403f2e2333d2ce56c30ba7b2c721245ad","observation_id":"63933835-d9dd-44a5-b42f-57788a1a3070","resolution":{"observed_at":"2026-08-15T21:08:56.923655Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-08-17T08:58:04.443078Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-08-07T14:31:47.967756Z","title":"Arithmetic control of llms for diverse user preferences: Directional preference alignment with multi-objective rewards","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21537","last_updated":"2025-05-24T09:07:13Z","snapshot_observed_at":"2026-08-13T18:49:31.787472Z","submitted_at":"2025-05-24T09:07:13Z","title":"OpenReview Should be Protected and Leveraged as a Community Asset for Research in the Era of Large Language Models","version":1},"reference_index":151,"source":"arxiv_source","source_observed_at":"2026-08-07T14:31:47.967756Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2505.21537"},"observation_digest":"sha256:d7faba1c5ceafe44830f55a4beefc905d6304e92746fd76efb56a512108fef7f","observation_id":"2c8814b4-ac09-42a0-8df4-2e87ae403b07","resolution":{"observed_at":"2026-08-07T14:31:47.967756Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-08-17T08:58:04.443078Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-08-06T18:10:18.299705Z","title":"Arithmetic control of llms for diverse user preferences: Directional preference alignment with multi-objective rewards","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09060","last_updated":"2025-07-15T17:48:41Z","snapshot_observed_at":"2026-08-17T19:44:51.189656Z","submitted_at":"2025-07-11T22:33:11Z","title":"CALMA: A Process for Deriving Context-aligned Axes for Language Model Alignment","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-06T18:10:18.299705Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2507.09060"},"observation_digest":"sha256:809bf9785a889fdb4c2a9ba9742a1196ee280f84557f24359d4ec838aef3c019","observation_id":"e7c8015e-3c13-4bf1-bfbc-7df13e0bfdea","resolution":{"observed_at":"2026-08-06T18:10:18.299705Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-08-17T08:58:04.443078Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":"2402.18571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Arithmetic control of LLMs for diverse user preferences: Directional preference alignment with multi-objective rewards","venue":null,"work_id":"891ab6ad-8181-4659-9ba7-8096330036f5","year":2024},"citing_paper":{"arxiv_id":"2507.17746","last_updated":"2025-10-03T01:55:55Z","snapshot_observed_at":"2026-08-12T14:23:22.826603Z","submitted_at":"2025-07-23T17:57:55Z","title":"Rubrics as Rewards: Reinforcement Learning Beyond Verifiable Domains","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-13T06:07:56.678339Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2507.17746"},"observation_digest":"sha256:2ddf12a31e5d102e4ba9e93253524391ecf2c65ae0528e2d48cc26c50b76456d","observation_id":"455ce814-39ff-4b23-ac00-f1f8cb8a6233","resolution":{"observed_at":"2026-05-13T06:07:56.859644Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-08-17T08:58:04.443078Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":"2402.18571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Arithmetic control of LLMs for diverse user preferences: Directional preference alignment with multi-objective rewards","venue":null,"work_id":"891ab6ad-8181-4659-9ba7-8096330036f5","year":2024},"citing_paper":{"arxiv_id":"2604.24536","last_updated":"2026-04-27T14:33:45Z","snapshot_observed_at":"2026-08-12T16:48:28.219032Z","submitted_at":"2026-04-27T14:33:45Z","title":"Generating Place-Based Compromises Between Two Points of View","version":1},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-05-08T03:36:31.695964Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2604.24536"},"observation_digest":"sha256:28e2f9565b7781c0dc2d72e140884ad9bb9bacdfae46e001abf6d74ac2f073f5","observation_id":"6f690dbf-7c73-41e2-8399-c55da0741e86","resolution":{"observed_at":"2026-05-11T22:01:12.114202Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-08-17T08:58:04.443078Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":"2402.18571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Arithmetic control of LLMs for diverse user preferences: Directional preference alignment with multi-objective rewards","venue":null,"work_id":"891ab6ad-8181-4659-9ba7-8096330036f5","year":2024},"citing_paper":{"arxiv_id":"2605.06987","last_updated":"2026-05-07T22:05:23Z","snapshot_observed_at":"2026-07-31T05:30:22.249144Z","submitted_at":"2026-05-07T22:05:23Z","title":"Response Time Enhances Alignment with Heterogeneous Preferences","version":1},"reference_index":164,"source":"arxiv_source","source_observed_at":"2026-05-11T01:04:26.288913Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2605.06987"},"observation_digest":"sha256:30ed3de207eba7527e15532e5ec5b40fe21349c57c7e361a28eb849e27f96762","observation_id":"b3524346-517f-4318-8abe-cf6999948d15","resolution":{"observed_at":"2026-05-11T04:45:59.808702Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-08-17T08:58:04.443078Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":"2402.18571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Arithmetic control of LLMs for diverse user preferences: Directional preference alignment with multi-objective rewards","venue":null,"work_id":"891ab6ad-8181-4659-9ba7-8096330036f5","year":2024},"citing_paper":{"arxiv_id":"2605.07162","last_updated":"2026-05-08T02:47:30Z","snapshot_observed_at":"2026-08-16T23:49:42.481692Z","submitted_at":"2026-05-08T02:47:30Z","title":"CLIPer: Tailoring Diverse User Preference via Classifier-Guided Inference-Time Personalization","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-05-11T02:16:28.593349Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2605.07162"},"observation_digest":"sha256:fc33dde75bb9c30642ca140d983f3c5a80383c8d3caa4c0a05c2d0786e0c064b","observation_id":"e1eb872b-f9b3-4eca-b37c-7cdbd9eec21d","resolution":{"observed_at":"2026-05-11T03:50:54.299657Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-08-17T08:58:04.443078Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":"2402.18571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Arithmetic control of LLMs for diverse user preferences: Directional preference alignment with multi-objective rewards","venue":null,"work_id":"891ab6ad-8181-4659-9ba7-8096330036f5","year":2024},"citing_paper":{"arxiv_id":"2605.11679","last_updated":"2026-05-13T09:28:34Z","snapshot_observed_at":"2026-08-21T21:23:16.145234Z","submitted_at":"2026-05-12T07:38:59Z","title":"Explaining and Breaking the Safety-Helpfulness Ceiling via Preference Dimensional Expansion","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-13T01:03:10.263663Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2605.11679"},"observation_digest":"sha256:4926262b665e66ac34f1a78deef0cc1f1baa49b277e6fc8738eefb3719fe5a1a","observation_id":"f5a28b24-336f-4925-9cc1-7de381f28271","resolution":{"observed_at":"2026-05-13T01:57:06.526224Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-08-17T08:58:04.443078Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":"2402.18571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Arithmetic control of LLMs for diverse user preferences: Directional preference alignment with multi-objective rewards","venue":null,"work_id":"891ab6ad-8181-4659-9ba7-8096330036f5","year":2024},"citing_paper":{"arxiv_id":"2605.11679","last_updated":"2026-05-13T09:28:34Z","snapshot_observed_at":"2026-08-21T21:23:16.145234Z","submitted_at":"2026-05-12T07:38:59Z","title":"Explaining and Breaking the Safety-Helpfulness Ceiling via Preference Dimensional Expansion","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-14T21:12:06.989077Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2605.11679"},"observation_digest":"sha256:02373f1fb6557a630b5750d1779309da63f2e1e5b5be035dc24f1277d9267164","observation_id":"7675384e-1375-4dae-9231-c096f722abec","resolution":{"observed_at":"2026-05-14T21:12:58.887870Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-08-17T08:58:04.443078Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":"2402.18571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Arithmetic control of LLMs for diverse user preferences: Directional preference alignment with multi-objective rewards","venue":null,"work_id":"891ab6ad-8181-4659-9ba7-8096330036f5","year":2024},"citing_paper":{"arxiv_id":"2605.19330","last_updated":"2026-05-19T04:07:41Z","snapshot_observed_at":"2026-08-16T23:42:07.834655Z","submitted_at":"2026-05-19T04:07:41Z","title":"MOCHA: Multi-Objective Chebyshev Annealing for Agent Skill Optimization","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-20T06:09:56.684622Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2605.19330"},"observation_digest":"sha256:749f99a2807b2842c4f6dab1d3cd80a74fa11931ba58905ce03950d9be7eb996","observation_id":"1de03b3c-9dc8-4e29-bbd7-4c767030b6f2","resolution":{"observed_at":"2026-05-20T06:13:05.313131Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-08-17T08:58:04.443078Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":"2402.18571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Arithmetic control of LLMs for diverse user preferences: Directional preference alignment with multi-objective rewards","venue":null,"work_id":"891ab6ad-8181-4659-9ba7-8096330036f5","year":2024},"citing_paper":{"arxiv_id":"2605.20408","last_updated":"2026-05-19T19:04:47Z","snapshot_observed_at":"2026-07-06T23:30:58.549353Z","submitted_at":"2026-05-19T19:04:47Z","title":"Spectral Souping: A Unified Framework for Online Preference Alignment","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-21T07:54:56.356555Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2605.20408"},"observation_digest":"sha256:8e203623ce65b0ce25a6684587675cbd7c2e045d8214f490d88b73fb1eb875ae","observation_id":"bb1526a0-e8e7-49a9-8bfd-b38fd5a1973e","resolution":{"observed_at":"2026-05-21T07:59:50.979793Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-08-17T08:58:04.443078Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-07-11T13:53:36.775836Z","title":"arXiv preprint arXiv:2402.18571 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.04763","last_updated":"2026-07-26T14:17:18Z","snapshot_observed_at":"2026-08-15T22:59:02.741838Z","submitted_at":"2026-07-06T07:56:53Z","title":"Multi-Turn On-Policy Distillation with Prefix Replay","version":1},"reference_index":148,"source":"arxiv_source","source_observed_at":"2026-07-11T13:53:36.775836Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2607.04763"},"observation_digest":"sha256:d821628b084291c49aa1519160f8526bdfcb17c965fff5f9e2ad42ce5aa7b190","observation_id":"dfaca053-4c67-49e6-b91a-07cf2c2eab44","resolution":{"observed_at":"2026-07-11T13:53:36.775836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-08-17T08:58:04.443078Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-08-02T08:40:48.709771Z","title":"arXiv preprint arXiv:2402.18571 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.04763","last_updated":"2026-07-26T14:17:18Z","snapshot_observed_at":"2026-08-15T22:59:02.741838Z","submitted_at":"2026-07-06T07:56:53Z","title":"Multi-Turn On-Policy Distillation with Prefix Replay","version":3},"reference_index":149,"source":"arxiv_source","source_observed_at":"2026-08-02T08:40:48.709771Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2607.04763"},"observation_digest":"sha256:6ab674569945828531ff9c586a835262b0fea8a17967f6762d08beca24957e94","observation_id":"364a638b-056d-4e49-bf1e-b4ceb9c665a2","resolution":{"observed_at":"2026-08-02T08:40:48.709771Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-08-17T08:58:04.443078Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-08-15T14:33:57.520968Z","title":"arXiv preprint arXiv:2402.18571 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.07419","last_updated":"2026-08-07T17:05:10Z","snapshot_observed_at":"2026-08-19T11:38:58.290988Z","submitted_at":"2026-08-07T17:05:10Z","title":"Beyond Post-Hoc Temperature Scaling: Bilevel Optimization for LLM Calibration","version":1},"reference_index":162,"source":"arxiv_source","source_observed_at":"2026-08-15T14:33:57.520968Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2608.07419"},"observation_digest":"sha256:b38ef920825a3fd991fd27fa88dd58a6416d48630cf147ec4b4367f4defcdaaf","observation_id":"f5a5977d-de80-4375-9be3-5169aed0a61b","resolution":{"observed_at":"2026-08-15T14:33:57.520968Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2402.18571/citation-record","integrity":"/paper/2402.18571/integrity","json":"/paper/2402.18571/citation-record.json","paper":"/paper/2402.18571"},"outbound":[],"paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","latest_version":3,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-17T08:58:04.443078Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"thesis":"As of 23 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 19 inbound Pith citation observations for arXiv:2402.18571."}