{"as_of":"2026-08-11T17:52:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:21cf89ccb9d0d4704684be63d8427581d6eb19faa656c1c5d60ecd6d8a0bd366","coverage":[{"denominator":42,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":42,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T10:19:36.208057Z","state":"measured"},{"denominator":42,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":42,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.05748/citation-record","integrity":"/paper/2506.05748/integrity","json":"/paper/2506.05748/citation-record.json","paper":"/paper/2506.05748"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:41.094546Z","title":"Each category of algorithms presents unique benefits and constraints, rendering their integrated application beneficial in real -world scenarios","venue":null,"work_id":"2748334d-5e30-4865-96dc-b64cf211c62c","year":null},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:32.904699Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:02b390ff29953d2993929d70c179b73a6e513c07eb43cb076b476ae68845c9cf","observation_id":"bfc05f2a-2cd1-423e-b517-028202c5982b","resolution":{"observed_at":"2026-08-07T10:19:41.116682Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:41.052485Z","title":"LLM-as- a-Judge","venue":null,"work_id":"89f6d4e1-5412-493d-8142-180185779c60","year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:32.952718Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:22b8d724690193e558c2fb17a22dd588d9b5ddf774ec0abd3fee2b4f83cf4098","observation_id":"4bf31566-4f24-4634-8570-42b5186bb4d4","resolution":{"observed_at":"2026-08-07T10:19:41.070207Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:40.828779Z","title":"score\" field in [-1, 1] and a short","venue":null,"work_id":"a63fccda-ba97-4375-a7d9-9473386417b5","year":null},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:33.128347Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:9174a9a42f923d56f8fa409cba273783bd125c434f91c45681a4def99d289dd3","observation_id":"c4e8b6e8-0188-4312-a89d-4c30f929e8c7","resolution":{"observed_at":"2026-08-07T10:19:40.876633Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:40.574827Z","title":"𝑏𝑒𝑡𝑡𝑒𝑟\":","venue":null,"work_id":"b0cdd5f6-83a0-4745-9ca8-e099c22cf288","year":null},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:33.304988Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:37959c9b1eacb694a1391e86fef4ac0258d3094d0c01f5cbc0559710504fd596","observation_id":"9c517a11-173d-4123-849e-a78414a33087","resolution":{"observed_at":"2026-08-07T10:19:40.636409Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:40.686982Z","title":"be funnier","venue":null,"work_id":"e8cccb6c-e444-4fd8-bb68-fd20df5f9d90","year":null},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:33.229860Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:1af41f66db0c138872876fd9ecd578d431c9e805b209c3cda550f4eddf31e4de","observation_id":"0394a0ad-67b0-4824-8573-f5df187a27c4","resolution":{"observed_at":"2026-08-07T10:19:40.765243Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:40.337606Z","title":"plug -and-play","venue":null,"work_id":"bf3826be-504e-4c39-809c-960d49d29dd8","year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:33.434466Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:fc3414e302851cd80941e15a09da0586072a1f221b356053861faa7f64cfa6e5","observation_id":"21804ec0-a5c9-4bad-a29b-1e5668030674","resolution":{"observed_at":"2026-08-07T10:19:40.366764Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:40.443151Z","title":"Which answer is better? Return ‘A’ or ‘B’ only","venue":null,"work_id":"83183cee-ac06-4508-a0c8-522878a18241","year":null},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:33.363323Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:e0165cdbdcf31ee70a86301cc1a8fc6730f0c05e27aba250e947de1002081c7e","observation_id":"ed1ee598-6ad1-4a86-b403-bab2b8c2cefc","resolution":{"observed_at":"2026-08-07T10:19:40.499258Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:39.824666Z","title":null,"venue":null,"work_id":"37dab451-4c59-4f9f-acad-d2418ee6a9f2","year":null},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:33.586566Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:d413f506a624ca5ce5a9c305895778c00d36bf14394f4bdd5ea4659b3d7ca331","observation_id":"2966fd8d-7156-4a75-887e-f9bfc63802e3","resolution":{"observed_at":"2026-08-07T10:19:39.929275Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:40.059753Z","title":"A” or “B","venue":null,"work_id":"32d8cba5-112a-4c85-a270-d34cc1b95863","year":null},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:33.507669Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:06ffb27378186e6eb0494d9d09d4a909d338f89cec598811e71f625410210f81","observation_id":"07f5747b-7a14-49e1-9a5b-c3a911624fd7","resolution":{"observed_at":"2026-08-07T10:19:40.160618Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"4287.34542","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:37.443445Z","title":"Online and Offline Reinforcement Learning by Planning with a Learned Model,","venue":null,"work_id":"2969fcd4-5738-480e-af7e-41c0e7af41d0","year":2021},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:34.462170Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:6201f1c8da5bf1ed38dbe32d6f6f976f6a8b7259cf73f692f1319241757e0472","observation_id":"008d4bbd-9253-4423-838d-373d19c01c03","resolution":{"observed_at":"2026-08-07T10:19:37.491599Z","resolver_source":"raw_fallback","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:39.505765Z","title":"Direct Preference Optimization: Your Language Model is Secretly a Reward Model Oral,","venue":null,"work_id":"91e86425-f9a5-47c1-affe-b7f7cf0dea0e","year":2023},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:33.659848Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:57ddf646ec4566796caa952397b85f542def7ae6d5455725526305235c21cf4f","observation_id":"8bb93105-9cad-4e16-aae9-7de035cd3255","resolution":{"observed_at":"2026-08-07T10:19:39.631521Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.48550/arxiv.2312.14925","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"A Survey of Reinforcement Learning from Human Feedback,","venue":"arXiv (Cornell University)","work_id":"133ebc6d-ab80-4282-949f-5b64dcbc1c67","year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:33.727935Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:b095804528ce21bea83a757cf9ccaa798a22463d821cff1d656948488786dea8","observation_id":"4230bc2a-447e-413d-b868-637db43f9f7d","resolution":{"observed_at":"2026-08-07T10:19:36.981902Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-03T22:08:26.692731+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T22:08:26.692731+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-07T10:19:33.848792Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:33.848792Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:815e60ac33ee0af4f788f56dd643b864d874337c13f8336ad4f0ac86a40d47fc","observation_id":"52201413-e62d-4376-99d4-f540ac16dbc2","resolution":{"observed_at":"2026-08-07T10:19:33.848792Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:33.926564Z","title":"Security and Privacy Challenges of Large Language Models: A Survey,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:33.926564Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:054465cf808601a3e4c82fcd80c5276dc44bd60f491f327a309e52aca2807d8e","observation_id":"650048c6-1110-4069-93d1-b38e3de8259a","resolution":{"observed_at":"2026-08-07T10:19:33.926564Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:34.046186Z","title":"A Survey on Hallucination in Large Language Models: Principles, Taxonomy, Challenges, and Open Questions,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:34.046186Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:6a516ba8d4870efe93bc398822c313544f6fb0554bc588f19da1339101e0a691","observation_id":"9f9e4aee-6743-4532-8a83-70b6c4ab6cb9","resolution":{"observed_at":"2026-08-07T10:19:34.046186Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-07T10:19:34.135241Z","title":"Qwen2.5 Technical Report,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:34.135241Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:54623994d062a83aa231d7d42e16f86f4707129921ac37b7fce9fee1d7e4f698","observation_id":"9a466f73-917c-48b2-9f60-e20479759f8d","resolution":{"observed_at":"2026-08-07T10:19:34.135241Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:39.258894Z","title":"Self-rewarding language models,","venue":null,"work_id":"9b36bab2-5e64-4913-995a-fba057e96197","year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:34.252515Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:159f734303feb43121763c97d22c03fdd12f46e65906fe7b59cc3eab34a4c4fe","observation_id":"008311f0-f88e-4e7a-a34e-34c079fed166","resolution":{"observed_at":"2026-08-07T10:19:39.400464Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:39.014875Z","title":"RLAIF: Scaling Reinforcement Learning from Human Feedback with AI Feedback,","venue":null,"work_id":"68d849ae-d718-4cbe-a5c9-3188122f2161","year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:34.294065Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:5440d6a7741a0a3de3022d94479e8cef94576f3f4461d67942f830ae91c4feb3","observation_id":"d4f62995-bf80-46ad-bb75-e2a1664aeb75","resolution":{"observed_at":"2026-08-07T10:19:39.116623Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:34.415108Z","title":"Evaluating Text -to-Visual Generation with Image -to-Text Generation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:34.415108Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:03ac90bbde812f5b30072ab1cd4d35542fd83bb5ada98049348286ebe7c2db78","observation_id":"3cdbe7f3-4c87-41ac-8be6-f70283a8170b","resolution":{"observed_at":"2026-08-07T10:19:34.415108Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:40.959168Z","title":"more proficient","venue":null,"work_id":"8b29270b-1f5a-4763-9e3a-6429ee71deaa","year":null},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:33.043123Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:2fbf578beee5b174c4e5be7e0002de55a671d136b9231016d406906fd12d104f","observation_id":"2e689453-8921-4ec3-85a3-8d38b9adadda","resolution":{"observed_at":"2026-08-07T10:19:41.034025Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"0270.36022","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:37.243015Z","title":"Training Language Models to Follow Instructions with Human Feedback,","venue":null,"work_id":"ff542819-e434-4c4e-9c5d-4506384430fb","year":2022},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:34.523346Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:53061e3491efc888d0a327bcb188c643a898f47fd2824673e04aa99037e311ac","observation_id":"50ef29fd-e9e9-4a45-bdb5-a57243e42712","resolution":{"observed_at":"2026-08-07T10:19:37.289842Z","resolver_source":"raw_fallback","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.10400","last_updated":"2025-02-24T08:57:10Z","snapshot_observed_at":"2026-08-10T14:49:48.013880Z","submitted_at":"2024-12-05T16:10:42Z","title":"Reinforcement Learning Enhanced LLMs: A Survey","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.10400","snapshot_observed_at":"2026-08-07T10:19:34.611764Z","title":"Reinforcement Learning Enhanced LLMs: A Survey,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:34.611764Z"},"links":{"cited_paper":"/paper/2412.10400","citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:b3c5be7124258234518acf1911ca8a480877e05988622ddb68958b0f22384250","observation_id":"b3f198b2-d25f-4a79-afe5-066bbddecdb4","resolution":{"observed_at":"2026-08-07T10:19:34.611764Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:34.727234Z","title":"Survey on Large Language Model -Enhanced Reinforcement Learning: Concept, Taxonomy, and Methods,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:34.727234Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:8ca74c285e2f109c6275911cf22d4d573b6488d1d8ee2d9cd666bc9f64db7c27","observation_id":"27d2466d-933e-4b88-ad92-f2b3074f42d3","resolution":{"observed_at":"2026-08-07T10:19:34.727234Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.17376","last_updated":"2025-04-24T08:50:01Z","snapshot_observed_at":"2026-08-10T05:50:37.485708Z","submitted_at":"2025-04-24T08:50:01Z","title":"On-Device Qwen2.5: Efficient LLM Inference with Model Compression and Hardware Acceleration","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.17376","snapshot_observed_at":"2026-08-07T10:19:34.793339Z","title":"On-Device Qwen2.5: Efficient LLM Inference with Model Compression and Hardware Acceleration,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:34.793339Z"},"links":{"cited_paper":"/paper/2504.17376","citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:f46ef65e82073dca15e9f9b4da67c8316592a639f5924a9fe57e8958eed91785","observation_id":"74ae2aea-c188-4db3-8f8c-e0aa6aeca16a","resolution":{"observed_at":"2026-08-07T10:19:34.793339Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.02554","last_updated":"2023-04-05T16:17:32Z","snapshot_observed_at":"2026-08-10T04:48:52.341807Z","submitted_at":"2023-04-05T16:17:32Z","title":"Human-like Summarization Evaluation with ChatGPT","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.02554","snapshot_observed_at":"2026-08-07T10:19:34.827221Z","title":"Human -like Summarization Evaluation with ChatGPT,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:34.827221Z"},"links":{"cited_paper":"/paper/2304.02554","citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:32133f24cd7b206a207b5ac915150005d9208afb439c83e946c0584998e5deff","observation_id":"7ed344d8-cc99-4521-8a1f-236530c45dda","resolution":{"observed_at":"2026-08-07T10:19:34.827221Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02736","last_updated":"2024-10-04T03:57:47Z","snapshot_observed_at":"2026-08-01T08:21:19.528254Z","submitted_at":"2024-10-03T17:53:30Z","title":"Justice or Prejudice? Quantifying Biases in LLM-as-a-Judge","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02736","snapshot_observed_at":"2026-08-07T10:19:34.917509Z","title":"Justice or Prejudice? Quantifying Biases in LLM-as-a-Judge,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:34.917509Z"},"links":{"cited_paper":"/paper/2410.02736","citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:743365b485dee1f4def5657d6493f63c4776137ace90978444ebf8a6ec0a95cb","observation_id":"2d96a1e9-2f2a-4270-9eba-9578b9740e94","resolution":{"observed_at":"2026-08-07T10:19:34.917509Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.11239","last_updated":"2024-10-02T11:07:14Z","snapshot_observed_at":"2026-08-03T08:44:22.056144Z","submitted_at":"2024-09-17T14:40:02Z","title":"LLM-as-a-Judge & Reward Model: What They Can and Cannot Do","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.11239","snapshot_observed_at":"2026-08-07T10:19:34.977810Z","title":"LLM-as-a-Judge & Reward Model: What They Can and Cannot Do,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:34.977810Z"},"links":{"cited_paper":"/paper/2409.11239","citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:73bc24b8ad40290428827fe7ee044e90cc1decaaa795b3c756f71a2211fb16a0","observation_id":"be93dd90-2e75-4460-9faa-018aaa2009e0","resolution":{"observed_at":"2026-08-07T10:19:34.977810Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:38.789565Z","title":"RLAIF vs. RLHF: scaling reinforcement learning from human feedback with AI feedback,","venue":null,"work_id":"138a76fd-034e-4b0d-8986-95656d8113f4","year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:35.034720Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:d5cc985fb386bff1a513bab98e3c38bb6c56e08704790653ac10228962aa27bf","observation_id":"b0a10ede-dbbe-42c8-b22f-a4d359ab22be","resolution":{"observed_at":"2026-08-07T10:19:38.869784Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:35.096249Z","title":"Large Language Models Can Self -Improve,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:35.096249Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:06a4bbc19068a6a3890b6cb928fbecc7fdb575bb5e260becb3094ca03d617509","observation_id":"fd153a62-f658-4573-8f88-c7e1ac52c102","resolution":{"observed_at":"2026-08-07T10:19:35.096249Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:38.612071Z","title":"Advancing Large Language Model Attribution through Self-Improving,","venue":null,"work_id":"1f178dff-abcb-45ac-8293-1c0745091597","year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:35.180761Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:e6331f3e893965a260024ba01aea084bccc7b4ce57737e08f48984c7a0e3b795","observation_id":"8bac4ca2-9414-4a58-9b45-926eb8f81762","resolution":{"observed_at":"2026-08-07T10:19:38.691907Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.13006","last_updated":"2025-03-30T17:59:47Z","snapshot_observed_at":"2026-07-06T19:05:05.530925Z","submitted_at":"2024-08-23T11:49:01Z","title":"Systematic Evaluation of LLM-as-a-Judge in LLM Alignment Tasks: Explainable Metrics and Diverse Prompt Templates","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.13006","snapshot_observed_at":"2026-08-07T10:19:35.386305Z","title":"Systematic Evaluation of LLM -as-a-Judge in LLM Alignment Tasks: Explainable Metrics and Diverse Prompt Templates,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:35.386305Z"},"links":{"cited_paper":"/paper/2408.13006","citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:cad7bac6d6ef32674a6b888a336a0973539c440f41a8940e026751fb739c60a8","observation_id":"7c9c9731-28a4-45d0-914a-91332507933e","resolution":{"observed_at":"2026-08-07T10:19:35.386305Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:35.466565Z","title":"Can LLM be a Personalized Judge?,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:35.466565Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:635fb177e055fc9dc50f65a213b1f5a51c24ac1775e6763b4cd17685522cbca0","observation_id":"3ed7257e-2d9d-42ea-b02f-56b633518d3b","resolution":{"observed_at":"2026-08-07T10:19:35.466565Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:38.418751Z","title":"ReST-MCTS*: LLM Self- Training via Process Reward Guided Tree Search,","venue":null,"work_id":"3b3afa43-3a3e-4c84-af76-13f13d1966f6","year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:35.548712Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:91a65ae608aec15b96fadee0301300d5bb3fedf229d3354b311c73d1795eb3b7","observation_id":"485880b4-b338-414d-b015-92fa35054ec6","resolution":{"observed_at":"2026-08-07T10:19:38.522228Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:38.126489Z","title":"Self-Play Preference Optimization for Language Model Alignment,","venue":null,"work_id":"f8ef14e2-593e-4225-b94d-8189bf0f7cce","year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:35.634935Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:589673158fef8a2522ba18e03c4fa3c0a46dd8ef41b8fbe0e3dd8b55babfaed1","observation_id":"56da1a36-06d1-4eee-a34c-2ce65173a5a5","resolution":{"observed_at":"2026-08-07T10:19:38.280897Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:37.666571Z","title":"Training language models to follow instructions with human feedback,","venue":null,"work_id":"8a149f52-5c30-44d9-9978-3397db30c6dd","year":2022},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:35.826010Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:31aba71e32ef85ae05d9f2f49a95357cd7d339fc5d981681973061e9cddec445","observation_id":"599d70aa-640b-4ce5-836e-8eaff13c67e2","resolution":{"observed_at":"2026-08-07T10:19:37.785037Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.08073","last_updated":"2022-12-15T06:19:23Z","snapshot_observed_at":"2026-08-02T04:53:58.766070Z","submitted_at":"2022-12-15T06:19:23Z","title":"Constitutional AI: Harmlessness from AI Feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.08073","snapshot_observed_at":"2026-08-07T10:19:35.928792Z","title":"Constitutional AI: Harmlessness from AI Feedback,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:35.928792Z"},"links":{"cited_paper":"/paper/2212.08073","citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:0683d151839519e409df05fed5c3bb07b43b108e9006ee50494b2976583ca83a","observation_id":"3bcf55f9-03af-4a07-bc39-bd4c46f401cb","resolution":{"observed_at":"2026-08-07T10:19:35.928792Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.00267","last_updated":"2024-09-03T14:01:54Z","snapshot_observed_at":"2026-07-06T16:13:07.384791Z","submitted_at":"2023-09-01T05:53:33Z","title":"RLAIF vs. RLHF: Scaling Reinforcement Learning from Human Feedback with AI Feedback","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.00267","snapshot_observed_at":"2026-08-07T10:19:35.995780Z","title":"RLAIF vs. RLHF: Scaling Reinforcement Learning from Human Feedback with AI Feedback,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:35.995780Z"},"links":{"cited_paper":"/paper/2309.00267","citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:84d2154876681183957e99bcf897b0f33405f46185aab73fd6b567d81cba9634","observation_id":"b18f9567-20f9-477a-9a02-cfd1064c7b01","resolution":{"observed_at":"2026-08-07T10:19:35.995780Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06180","last_updated":"2023-09-12T12:50:04Z","snapshot_observed_at":"2026-08-02T09:51:08.145755Z","submitted_at":"2023-09-12T12:50:04Z","title":"Efficient Memory Management for Large Language Model Serving with PagedAttention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.06180","snapshot_observed_at":"2026-08-07T10:19:36.068399Z","title":"Efficient Memory Management for Large Language Model Serving with PagedAttention,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:36.068399Z"},"links":{"cited_paper":"/paper/2309.06180","citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:a72629a3335236a87e8311447f1019b142b9d33f4ca32aa3239967abe5d8c1c8","observation_id":"99d79685-81e9-4773-abbd-48ea36bbf58a","resolution":{"observed_at":"2026-08-07T10:19:36.068399Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.13787","last_updated":"2024-06-08T16:40:12Z","snapshot_observed_at":"2026-08-02T18:11:57.036767Z","submitted_at":"2024-03-20T17:49:54Z","title":"RewardBench: Evaluating Reward Models for Language Modeling","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.13787","snapshot_observed_at":"2026-08-07T10:19:36.134562Z","title":"RewardBench: Evaluating Reward Models for Language Modeling,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:36.134562Z"},"links":{"cited_paper":"/paper/2403.13787","citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:be7c2459c6492aa83363b7a417e077476e23e400e8b11d4267809682638c873a","observation_id":"e97fd2f1-6d6e-44c7-9cc1-9fcf64545a28","resolution":{"observed_at":"2026-08-07T10:19:36.134562Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:37.569163Z","title":"Iterative Reasoning Preference Optimization,","venue":null,"work_id":"894f2473-c89e-4329-853f-0ec027bc90b3","year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:36.208057Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:42d4bc708a26b8b0c6ddb356c21716a8af5f191a6c42be6d9032003359b4f530","observation_id":"3be384a4-a619-4978-9f80-881a09551d7f","resolution":{"observed_at":"2026-08-07T10:19:37.604055Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:37.851865Z","title":"Available: https://neurips.cc/virtual/2024/108142","venue":null,"work_id":"db45c6e2-f833-49a3-8ffe-c54bcae052f2","year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:35.719817Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:318ac9b66790cc2e02823e6dca96211e04c66c57bb1f0f6cc2315f3e0e4b4281","observation_id":"16b69448-1360-4ae8-affb-5f829c56681e","resolution":{"observed_at":"2026-08-07T10:19:38.000462Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:35.287341Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":3836,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:35.287341Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:297516423981d03a4ceeec6b5f542fb3210133e2d051d61625d6d2752e8cb59e","observation_id":"b797fe01-8b85-4336-b0b0-c115757d42f8","resolution":{"observed_at":"2026-08-07T10:19:35.287341Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance"},"reference_resolution":{"displayed":42,"state_counts":{"malformed_identifier":0,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":20,"verified_exact":1,"verified_fuzzy":19},"total_outbound_references":42},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 42 of 42 outbound references and 0 inbound Pith citation observations for arXiv:2506.05748."}