{"as_of":"2026-08-10T20:45:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b661bed8e34808fd2ab59b074de42336acd2f394e60216f326d32e7886820204","coverage":[{"denominator":28,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":28,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-18T03:40:52.890485Z","state":"measured"},{"denominator":28,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":28,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2510.23868/citation-record","integrity":"/paper/2510.23868/integrity","json":"/paper/2510.23868/citation-record.json","paper":"/paper/2510.23868"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Goucher, and et al","venue":null,"work_id":"d3f67a6e-e6f7-4941-9649-c3158e5b401f","year":2024},"citing_paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA","version":5},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-18T03:40:52.890485Z"},"links":{"citing_paper":"/paper/2510.23868"},"observation_digest":"sha256:9c734d0982fd3dfff095a03c8b185b48a5fd56ddfe6360eb2ea51c90dfbdcd29","observation_id":"5475d7cc-5406-4fdd-bf33-e5f49e60b1ec","resolution":{"observed_at":"2026-05-18T03:42:23.569086Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Brown, Benjamin Mann, Nick Ryder, Melanie Subbiah, Jared Kaplan, and et al","venue":null,"work_id":"2b0fb1e9-a3c9-494c-b8c2-43534987b9e4","year":2020},"citing_paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA","version":5},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-18T03:40:52.890485Z"},"links":{"citing_paper":"/paper/2510.23868"},"observation_digest":"sha256:f8330d488660f23211d95fbea46ece0be9b29f774ce37e95b2544e49d42cfcf4","observation_id":"70b9a2aa-0b65-4c1e-84a3-67e1020ad88b","resolution":{"observed_at":"2026-05-18T03:42:23.587363Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Wainwright, and et al","venue":null,"work_id":"c825d1bc-c62a-474e-a3e8-edde7e49d26a","year":2022},"citing_paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA","version":5},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-18T03:40:52.890485Z"},"links":{"citing_paper":"/paper/2510.23868"},"observation_digest":"sha256:2b0276fec12fafe0e8e6ea14e600c78a3e72aec0498798cf28001aae215e44c5","observation_id":"a80d2e47-8470-4f75-a340-29f2b992c9e5","resolution":{"observed_at":"2026-05-18T03:42:23.576294Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning","venue":null,"work_id":"d15ca4e1-3a16-4bea-81d1-3e6171259499","year":2025},"citing_paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA","version":5},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-18T03:40:52.890485Z"},"links":{"citing_paper":"/paper/2510.23868"},"observation_digest":"sha256:ae16cddafe6ae1b3d270f36ae6216aeb902053bc5e78bb10318e7ce327830f10","observation_id":"f24d86b5-e79a-4c4c-b3a9-9cfde45988fb","resolution":{"observed_at":"2026-05-18T03:42:23.573606Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T10:06:10.239871Z","title":"Proximal policy optimization algorithms","venue":null,"work_id":"3fd006b3-ce9f-4777-887c-2e1b5edfdf04","year":2017},"citing_paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA","version":5},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-18T03:40:52.890485Z"},"links":{"citing_paper":"/paper/2510.23868"},"observation_digest":"sha256:ad8413d5dae4a8b7f2532872895e33765808370ecea253f2628d2cb0706fcc5e","observation_id":"8af92dcf-8fec-479b-ae55-6f10f92cfbbc","resolution":{"observed_at":"2026-05-18T03:42:23.616562Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Manning, and et al","venue":null,"work_id":"c98984f0-3a78-46d8-a67e-a5a33fe07ece","year":2024},"citing_paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA","version":5},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-18T03:40:52.890485Z"},"links":{"citing_paper":"/paper/2510.23868"},"observation_digest":"sha256:052b3ec3bfcaabcfc08b93c59f72e35e163d5671bf1f26d69339f942f0c902e0","observation_id":"fbfa0e2f-bc58-41bc-8d0b-5077c2ab436d","resolution":{"observed_at":"2026-05-18T03:42:23.619267Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Una: Unifying alignments of rlhf/ppo, dpo and kto by a generalized implicit reward function","venue":null,"work_id":"1655b55a-7a6c-4f37-842e-b2380e84c053","year":2025},"citing_paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA","version":5},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-18T03:40:52.890485Z"},"links":{"citing_paper":"/paper/2510.23868"},"observation_digest":"sha256:be21d662f81d9f089150b3aa8573dab8d32259d9ec59ea4074b82371216c564e","observation_id":"6d869c7c-55e2-4345-9f47-4416d4b42cbf","resolution":{"observed_at":"2026-05-18T03:42:23.625976Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T06:34:42.387207Z","title":"Qwen2.5: A party of foundation models, September 2024","venue":null,"work_id":"6c796970-6476-4183-926b-4a42316e5ea6","year":2024},"citing_paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA","version":5},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-18T03:40:52.890485Z"},"links":{"citing_paper":"/paper/2510.23868"},"observation_digest":"sha256:87913abacd6021563e259ddb7c3753276cd8e0d5fe7034d1966c14dd2c1f3606","observation_id":"a9315b28-417e-4dd9-9e3b-9355ee202440","resolution":{"observed_at":"2026-05-18T03:42:23.611175Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":"2110.14168","doi":"10.1002/j.1545-","metadata_source":"pith","pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Training Verifiers to Solve Math Word Problems","venue":"cs.LG","work_id":"acab1aa8-b4d6-40e0-a3ee-25341701dca2","year":2021},"citing_paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA","version":5},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-18T03:40:52.890485Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2510.23868"},"observation_digest":"sha256:5a5cb1ef695509f163ac1c7ab616d2c1f7ab64ddccddd921769927e7d7717ab1","observation_id":"0a34507d-a489-4b60-b5eb-442a2ef2c07e","resolution":{"observed_at":"2026-05-18T03:42:22.646514Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03874","last_updated":"2021-11-08T21:30:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-03-05T18:59:39Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","version":2},"cited_work":{"arxiv_id":"2103.03874","doi":"10.48550/arxiv.2103.03874","metadata_source":"pith","pith_arxiv_id":"2103.03874","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","venue":"cs.LG","work_id":"50652ac6-fb7c-4675-a2c2-159c241feb17","year":2021},"citing_paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA","version":5},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-18T03:40:52.890485Z"},"links":{"cited_paper":"/paper/2103.03874","citing_paper":"/paper/2510.23868"},"observation_digest":"sha256:28d796aa55ab36a0ff6e417db37eecc5730f141d1d8a0acdda1bb7d572be4781","observation_id":"22534931-0453-47a1-b402-c73b7e9531fe","resolution":{"observed_at":"2026-05-18T03:42:22.651882Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:22.649941+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:22.649941+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Dapo: An open-source llm reinforcement learning system at scale","venue":null,"work_id":"a649303f-8bd1-4069-bee0-11cf20e747f6","year":2025},"citing_paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA","version":5},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-18T03:40:52.890485Z"},"links":{"citing_paper":"/paper/2510.23868"},"observation_digest":"sha256:46a463e31edec171bbcb48d4af78b601b998b86a58af09e5c947f1f357ae7d3f","observation_id":"a670efb9-a995-464a-91f5-67b349f602ff","resolution":{"observed_at":"2026-05-18T03:42:23.571273Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Aime 2024 dataset","venue":null,"work_id":"a81c6eb2-604e-478c-8a9a-382cde76a719","year":2024},"citing_paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA","version":5},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-18T03:40:52.890485Z"},"links":{"citing_paper":"/paper/2510.23868"},"observation_digest":"sha256:73eb2280c8ed9ec211e6ad20ca9472ab912dfbd906bc0154f21729a014bb9204","observation_id":"bd2c5196-8d28-4861-bdc7-40b6de3b9fb1","resolution":{"observed_at":"2026-05-18T03:42:23.621485Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Qwen3 technical report","venue":null,"work_id":"f4a46f5c-ee36-48e7-b7cf-ca1ca677ad98","year":2025},"citing_paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA","version":5},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-18T03:40:52.890485Z"},"links":{"citing_paper":"/paper/2510.23868"},"observation_digest":"sha256:1d5b591a4eab9cb9404fe5315459a8da7e0ad96a22e591283d1a78185880f9a7","observation_id":"581df7ff-8b60-47dd-ab24-af9869760212","resolution":{"observed_at":"2026-05-18T03:42:23.608257Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Infinity instruct: Scaling instruction selection and synthesis to enhance language models","venue":null,"work_id":"806635eb-8d57-4343-b508-89f04b84a962","year":2025},"citing_paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA","version":5},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-18T03:40:52.890485Z"},"links":{"citing_paper":"/paper/2510.23868"},"observation_digest":"sha256:2c3d85698a949705bee18d3c88c6d39bf2c5640a63072d6e695cd9cb48420c1b","observation_id":"fbe7c208-8874-4ce6-bf38-fd75aea56924","resolution":{"observed_at":"2026-05-18T03:42:23.566960Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.01352","last_updated":"2026-03-02T20:56:17Z","snapshot_observed_at":"2026-08-04T07:18:25.941807Z","submitted_at":"2025-07-02T04:40:29Z","title":"Skywork-Reward-V2: Scaling Preference Data Curation via Human-AI Synergy","version":3},"cited_work":{"arxiv_id":"2507.01352","doi":"10.48550/arxiv.2507.01352","metadata_source":"pith","pith_arxiv_id":"2507.01352","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Skywork-Reward-V2: Scaling Preference Data Curation via Human-AI Synergy","venue":"cs.CL","work_id":"c44139eb-e611-43d2-80db-eaecb038fe32","year":2025},"citing_paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA","version":5},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-18T03:40:52.890485Z"},"links":{"cited_paper":"/paper/2507.01352","citing_paper":"/paper/2510.23868"},"observation_digest":"sha256:01ef518e39899eb8e0f1f04bdd67315ab4ba8f9a4792f720db4cfa11d577357a","observation_id":"679688b6-d791-4d72-8fe3-cb278c71fd94","resolution":{"observed_at":"2026-05-18T03:42:22.634732Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Truthfulqa: Measuring how models mimic human falsehoods","venue":null,"work_id":"66c287a5-ac6b-4acf-b207-6703a9d3274d","year":2022},"citing_paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA","version":5},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-18T03:40:52.890485Z"},"links":{"citing_paper":"/paper/2510.23868"},"observation_digest":"sha256:36800562d97f28d73f3ed807bfdaec91ac292ae574c31cd42d1d328892626b9a","observation_id":"29571d17-b23f-4e9b-b3b7-2756a9c338a9","resolution":{"observed_at":"2026-05-18T03:42:23.602839Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"BBQ: A hand-built bias benchmark for question answering","venue":null,"work_id":"64a82296-29ae-4d6d-b26a-bf3440b72fe4","year":2022},"citing_paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA","version":5},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-18T03:40:52.890485Z"},"links":{"citing_paper":"/paper/2510.23868"},"observation_digest":"sha256:6369e0b39e328c267e2c9cce99ba204c2cbec09c5b3e57a2e4c2bd42206f0cb8","observation_id":"2354c383-8420-46b1-8be7-8e70527df34f","resolution":{"observed_at":"2026-05-18T03:42:23.605539Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T14:26:21.198691Z","title":null,"venue":null,"work_id":"c69dcc4d-85af-4249-accc-cbae1a0e230c","year":null},"citing_paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA","version":5},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-18T03:40:52.890485Z"},"links":{"citing_paper":"/paper/2510.23868"},"observation_digest":"sha256:bc2422b4af16dc6c2b2b15eb0f98b699be31942970a3b678a507adc1591d6bd1","observation_id":"c8ac68d2-52b4-4fbd-aebe-932deb4fec50","resolution":{"observed_at":"2026-05-18T03:42:23.597278Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Program synthesis with large language models","venue":null,"work_id":"84115b1f-6510-49b3-a76a-dcd6309d7f8f","year":2021},"citing_paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA","version":5},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-18T03:40:52.890485Z"},"links":{"citing_paper":"/paper/2510.23868"},"observation_digest":"sha256:b6ed683d1ef366f5d64f6ee87af7d3e0aaab84e6f62d1f7d30fe422e9fe1982c","observation_id":"e83a78e2-4bae-451a-9a38-5ed9d6f300a4","resolution":{"observed_at":"2026-05-18T03:42:23.592212Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Think you have solved question answering? try arc, the ai2 reasoning challenge","venue":null,"work_id":"79b0e080-5a82-4d13-badb-9932509bc14d","year":2018},"citing_paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA","version":5},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-18T03:40:52.890485Z"},"links":{"citing_paper":"/paper/2510.23868"},"observation_digest":"sha256:1975d33b4f7d84cb388bd5c2cc526be954b1df6ec189ec27bbf96223549db7b0","observation_id":"6f7c27c2-842e-4d8a-92f2-a9f4aa62deae","resolution":{"observed_at":"2026-05-18T03:42:23.594901Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gender bias in coreference resolution","venue":null,"work_id":"b565af70-f365-49b0-9c7c-de2d3513dc10","year":2018},"citing_paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA","version":5},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-18T03:40:52.890485Z"},"links":{"citing_paper":"/paper/2510.23868"},"observation_digest":"sha256:f470394c9eb7e3bad26f4168fca21fb3ab32078529602e31a2e548776e9d7ee7","observation_id":"70500b9a-833f-4e3e-8305-4642481af6d3","resolution":{"observed_at":"2026-05-18T03:42:23.600326Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gpqa: A graduate-level google-proof q&a benchmark","venue":null,"work_id":"9a2a584e-cd84-40af-be33-270f5289b821","year":2023},"citing_paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA","version":5},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-18T03:40:52.890485Z"},"links":{"citing_paper":"/paper/2510.23868"},"observation_digest":"sha256:d1cec02a5e213ccc3d7884106f3389dde02bce56159eb508fd7e1ba1ac52a412","observation_id":"45dfdff3-12ae-40b3-b2af-04fa6bd14019","resolution":{"observed_at":"2026-05-18T03:42:23.613877Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Musr: Testing the limits of chain-of- thought with multistep soft reasoning","venue":null,"work_id":"d77cfdda-0a7b-4bcf-b2dd-54459bf8cc39","year":2024},"citing_paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA","version":5},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-18T03:40:52.890485Z"},"links":{"citing_paper":"/paper/2510.23868"},"observation_digest":"sha256:91f7caee2e3edfe4e2b8bb1f2894f09a59448fb49f80f9d501e5091fe5f3311a","observation_id":"8ef416a1-8dfb-4d75-9ce4-42234831eab7","resolution":{"observed_at":"2026-05-18T03:42:23.623736Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.04475","last_updated":"2025-03-10T09:27:03Z","snapshot_observed_at":"2026-07-06T17:56:23.317089Z","submitted_at":"2024-04-06T02:29:02Z","title":"Length-Controlled AlpacaEval: A Simple Way to Debias Automatic Evaluators","version":2},"cited_work":{"arxiv_id":"2404.04475","doi":"10.48550/arxiv.2404.04475","metadata_source":"pith","pith_arxiv_id":"2404.04475","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Length-Controlled AlpacaEval: A Simple Way to Debias Automatic Evaluators","venue":"cs.LG","work_id":"ef25adcf-addb-445e-b3b5-858eeb9883ca","year":2024},"citing_paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA","version":5},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-18T03:40:52.890485Z"},"links":{"cited_paper":"/paper/2404.04475","citing_paper":"/paper/2510.23868"},"observation_digest":"sha256:4539bc8f7690fc345807d0b8dc9fc3fb12b1bb79cda59833408f5b731bb3a165","observation_id":"b82b5ce8-8402-467b-8a2a-99d07555b580","resolution":{"observed_at":"2026-05-18T03:42:22.640237Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gonzalez, and Ion Sto- ica","venue":null,"work_id":"98acf129-b3b5-4f08-995b-05b7718cecdc","year":2024},"citing_paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA","version":5},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-18T03:40:52.890485Z"},"links":{"citing_paper":"/paper/2510.23868"},"observation_digest":"sha256:c671b81615a60c2f4605174cb2453f96ed7bf1e3c8522c12b2ab90ef3529f527","observation_id":"c4c17970-b47f-4203-b23d-fabd9a451479","resolution":{"observed_at":"2026-05-18T03:42:23.584709Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:44:43.389356Z","title":"Rank analysis of incomplete block designs: I","venue":null,"work_id":"7c92dc1c-5c6b-42d3-a3a8-1b5b19d220da","year":1952},"citing_paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA","version":5},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-18T03:40:52.890485Z"},"links":{"citing_paper":"/paper/2510.23868"},"observation_digest":"sha256:6a4fcf285d8810bf4deddf8d338ec08e46e87a5beefc7bd730291e9093e87fad","observation_id":"c2e09a9f-d606-409d-bbef-03632dc5352d","resolution":{"observed_at":"2026-05-18T03:42:23.579004Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Deepseekmath: Pushing the limits of mathematical reasoning in open language models","venue":null,"work_id":"79c6a0ce-98d1-4939-8dc6-ed0e04fe0b20","year":2024},"citing_paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA","version":5},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-18T03:40:52.890485Z"},"links":{"citing_paper":"/paper/2510.23868"},"observation_digest":"sha256:2bed00943454e2542a9929243fd3de38babfa3d31dc681809548209d4a43a764","observation_id":"f45f6251-a348-40ff-a91e-764fbc808b0f","resolution":{"observed_at":"2026-05-18T03:42:23.581687Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"aha moments","venue":null,"work_id":"17fab3f3-c824-4a18-81c4-3a1598a7bdba","year":2025},"citing_paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA","version":5},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-18T03:40:52.890485Z"},"links":{"citing_paper":"/paper/2510.23868"},"observation_digest":"sha256:1834bd39594fd6681329116e07e7b2ce725697121f5b6e683ecbfe89e2f2fce9","observation_id":"92474b42-46e9-442a-97e2-266a21f9336e","resolution":{"observed_at":"2026-05-18T03:42:23.589723Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2510.23868","last_updated":"2026-05-13T23:20:59Z","latest_version":5,"primary_category":"cs.LG","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T21:18:19Z","title":"GIFT: Group-Relative Implicit Fine-Tuning Integrates GRPO with DPO and UNA"},"reference_resolution":{"displayed":28,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":1,"verified_exact":4,"verified_fuzzy":23},"total_outbound_references":28},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 28 of 28 outbound references and 0 inbound Pith citation observations for arXiv:2510.23868."}