{"as_of":"2026-08-10T15:59:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:79e5b1a31a0a082874f2ccfe2146a91af7c8d41168109e98e7deafd598a12ba0","coverage":[{"denominator":61,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":61,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T17:32:34.292050Z","state":"measured"},{"denominator":72,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":72,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":11,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":11,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T20:21:21.808023Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-07T12:53:50.345403Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.10535","snapshot_observed_at":"2026-08-04T20:21:21.808023Z","title":"Jiang, S.; Huang, Z.; Luo, X.; and Sun, Y","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.08682","last_updated":"2025-09-10T15:22:00Z","snapshot_observed_at":"2026-08-08T09:57:01.058380Z","submitted_at":"2025-09-10T15:22:00Z","title":"Automatic Failure Attribution and Critical Step Prediction Method for Multi-Agent Systems Based on Causal Inference","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:21.808023Z"},"links":{"cited_paper":"/paper/2507.10535","citing_paper":"/paper/2509.08682"},"observation_digest":"sha256:ee3211f387d7379e2b49116512fab926084d67cc7890e89dd6ee3f17faa5a9ec","observation_id":"081fa9fd-8790-4a89-a31e-c299e3a060b9","resolution":{"observed_at":"2026-08-04T20:21:21.808023Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"cited_work":{"arxiv_id":"2507.10535","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.10535","snapshot_observed_at":"2026-07-07T12:53:50.345403Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-judge for coding tasks","venue":"cs.CL","work_id":"7e37bba3-99fb-4b09-af65-f199ccc4101e","year":2025},"citing_paper":{"arxiv_id":"2509.09936","last_updated":"2025-09-12T02:53:57Z","snapshot_observed_at":"2026-07-06T22:29:06.119014Z","submitted_at":"2025-09-12T02:53:57Z","title":"SciML Agents: Write the Solver, Not the Solution","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-18T17:57:51.444493Z"},"links":{"cited_paper":"/paper/2507.10535","citing_paper":"/paper/2509.09936"},"observation_digest":"sha256:68581685511d16e17c279c08f46fb3ced377a7b39efbc8a93e11f4a0fb465494","observation_id":"0b3b896f-db59-46f6-85d3-2005ee87ec9d","resolution":{"observed_at":"2026-05-18T18:01:43.843517Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"cited_work":{"arxiv_id":"2507.10535","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.10535","snapshot_observed_at":"2026-07-07T12:53:50.345403Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-judge for coding tasks","venue":"cs.CL","work_id":"7e37bba3-99fb-4b09-af65-f199ccc4101e","year":2025},"citing_paper":{"arxiv_id":"2604.16790","last_updated":"2026-04-18T02:35:05Z","snapshot_observed_at":"2026-07-06T23:04:01.558812Z","submitted_at":"2026-04-18T02:35:05Z","title":"Bias in the Loop: Auditing LLM-as-a-Judge for Software Engineering","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T07:29:03.994957Z"},"links":{"cited_paper":"/paper/2507.10535","citing_paper":"/paper/2604.16790"},"observation_digest":"sha256:6f1a11f79456b79a1999268301c4602222d44aa7fa974268a8a7f8125e14e8f7","observation_id":"9c5510e2-213a-4a3a-98d2-e90d1936e3c6","resolution":{"observed_at":"2026-05-10T07:32:00.313227Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"cited_work":{"arxiv_id":"2507.10535","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.10535","snapshot_observed_at":"2026-07-07T12:53:50.345403Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-judge for coding tasks","venue":"cs.CL","work_id":"7e37bba3-99fb-4b09-af65-f199ccc4101e","year":2025},"citing_paper":{"arxiv_id":"2604.27727","last_updated":"2026-04-30T11:20:22Z","snapshot_observed_at":"2026-07-06T23:13:11.623453Z","submitted_at":"2026-04-30T11:20:22Z","title":"LLM-as-a-Judge for Human-AI Co-Creation: A Reliability-Aware Evaluation Framework for Coding","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-07T08:39:55.256518Z"},"links":{"cited_paper":"/paper/2507.10535","citing_paper":"/paper/2604.27727"},"observation_digest":"sha256:f4f5243b996b4ab1fbaf7b1112100ce165dc67e7fbb7f266b707b2244c2217cc","observation_id":"f0897a24-f187-4a50-ba93-1a1b3bbfa912","resolution":{"observed_at":"2026-05-12T09:56:28.093044Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"cited_work":{"arxiv_id":"2507.10535","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.10535","snapshot_observed_at":"2026-07-07T12:53:50.345403Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-judge for coding tasks","venue":"cs.CL","work_id":"7e37bba3-99fb-4b09-af65-f199ccc4101e","year":2025},"citing_paper":{"arxiv_id":"2605.01474","last_updated":"2026-05-02T14:44:49Z","snapshot_observed_at":"2026-07-06T23:14:43.034336Z","submitted_at":"2026-05-02T14:44:49Z","title":"ReMedi: Reasoner for Medical Clinical Prediction","version":1},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-05-09T14:20:29.672994Z"},"links":{"cited_paper":"/paper/2507.10535","citing_paper":"/paper/2605.01474"},"observation_digest":"sha256:3f3b9c0329aa138717b9a435df5ce66e328c7d4da43c07c46a54e39bfe448ae3","observation_id":"5095cda1-e4e3-45de-acf0-126397d21c30","resolution":{"observed_at":"2026-05-11T17:01:05.917444Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"cited_work":{"arxiv_id":"2507.10535","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.10535","snapshot_observed_at":"2026-07-07T12:53:50.345403Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-judge for coding tasks","venue":"cs.CL","work_id":"7e37bba3-99fb-4b09-af65-f199ccc4101e","year":2025},"citing_paper":{"arxiv_id":"2605.02906","last_updated":"2026-07-31T03:14:41Z","snapshot_observed_at":"2026-08-05T23:10:40.522491Z","submitted_at":"2026-04-06T02:40:18Z","title":"OpsLLM: Construction of Large Language Model for Software Operations with Multi-stage Learning","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T20:07:07.548384Z"},"links":{"cited_paper":"/paper/2507.10535","citing_paper":"/paper/2605.02906"},"observation_digest":"sha256:942728c92bcf357c9825d846a3c32eb859c16617a31b70aacdf590f9c1f703bb","observation_id":"4ab9dd30-515b-4947-a9b1-41a95316c4bc","resolution":{"observed_at":"2026-05-10T22:15:48.632378Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"cited_work":{"arxiv_id":"2507.10535","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.10535","snapshot_observed_at":"2026-07-07T12:53:50.345403Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-judge for coding tasks","venue":"cs.CL","work_id":"7e37bba3-99fb-4b09-af65-f199ccc4101e","year":2025},"citing_paper":{"arxiv_id":"2605.02906","last_updated":"2026-07-31T03:14:41Z","snapshot_observed_at":"2026-08-05T23:10:40.522491Z","submitted_at":"2026-04-06T02:40:18Z","title":"OpsLLM: Construction of Large Language Model for Software Operations with Multi-stage Learning","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-13T06:25:16.650306Z"},"links":{"cited_paper":"/paper/2507.10535","citing_paper":"/paper/2605.02906"},"observation_digest":"sha256:4d023f77a33ee0a359ceebb91909a0d7742cbb0ad5ead0921cc47d58b94d0da6","observation_id":"5e398c30-299a-4459-83a9-72ed8fe09a01","resolution":{"observed_at":"2026-05-13T06:27:24.609554Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.10535","snapshot_observed_at":"2026-08-03T02:30:32.418305Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.02906","last_updated":"2026-07-31T03:14:41Z","snapshot_observed_at":"2026-08-05T23:10:40.522491Z","submitted_at":"2026-04-06T02:40:18Z","title":"OpsLLM: Construction of Large Language Model for Software Operations with Multi-stage Learning","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-03T02:30:32.418305Z"},"links":{"cited_paper":"/paper/2507.10535","citing_paper":"/paper/2605.02906"},"observation_digest":"sha256:e4b77af19da074e1c563c4bb78dabdd6f971882767d52707cd088b4e3bdaab3c","observation_id":"53877acc-d601-4396-a639-446a0da43499","resolution":{"observed_at":"2026-08-03T02:30:32.418305Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"cited_work":{"arxiv_id":"2507.10535","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.10535","snapshot_observed_at":"2026-07-07T12:53:50.345403Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-judge for coding tasks","venue":"cs.CL","work_id":"7e37bba3-99fb-4b09-af65-f199ccc4101e","year":2025},"citing_paper":{"arxiv_id":"2605.13139","last_updated":"2026-05-13T08:05:16Z","snapshot_observed_at":"2026-08-08T01:44:09.576347Z","submitted_at":"2026-05-13T08:05:16Z","title":"SWE-Cycle: Benchmarking Code Agents across the Complete Issue Resolution Cycle","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-14T18:34:39.997353Z"},"links":{"cited_paper":"/paper/2507.10535","citing_paper":"/paper/2605.13139"},"observation_digest":"sha256:b347f962dcae454c879bdd69bdcc5cd132e7ba339d25a262b2ae47bfd1a628f7","observation_id":"7e8a0566-9712-4552-86ea-45b3bb2a9231","resolution":{"observed_at":"2026-05-14T18:37:35.746653Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"cited_work":{"arxiv_id":"2507.10535","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.10535","snapshot_observed_at":"2026-07-07T12:53:50.345403Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-judge for coding tasks","venue":"cs.CL","work_id":"7e37bba3-99fb-4b09-af65-f199ccc4101e","year":2025},"citing_paper":{"arxiv_id":"2607.05391","last_updated":"2026-07-07T17:26:37Z","snapshot_observed_at":"2026-08-06T03:00:58.001418Z","submitted_at":"2026-07-06T17:59:35Z","title":"LLM-as-a-Verifier: A General-Purpose Verification Framework","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-07-07T12:47:29.552283Z"},"links":{"cited_paper":"/paper/2507.10535","citing_paper":"/paper/2607.05391"},"observation_digest":"sha256:c11840198f3631461662616a19e356a411fcdcaf14e51d9315721537f3a51670","observation_id":"c47fe0ad-dcbc-4e60-ba75-5ed87aa6e5fe","resolution":{"observed_at":"2026-07-07T12:53:50.347229Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.10535","snapshot_observed_at":"2026-07-11T07:02:51.850836Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.05391","last_updated":"2026-07-07T17:26:37Z","snapshot_observed_at":"2026-08-06T03:00:58.001418Z","submitted_at":"2026-07-06T17:59:35Z","title":"LLM-as-a-Verifier: A General-Purpose Verification Framework","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-07-11T07:02:51.850836Z"},"links":{"cited_paper":"/paper/2507.10535","citing_paper":"/paper/2607.05391"},"observation_digest":"sha256:b4041becceff7fdc0ef94853c5805043a279ac4a6274af09936638753033c8ba","observation_id":"b7a9f655-9f30-4eb0-99d8-c68ab6d5b659","resolution":{"observed_at":"2026-07-11T07:02:51.850836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2507.10535/citation-record","integrity":"/paper/2507.10535/integrity","json":"/paper/2507.10535/citation-record.json","paper":"/paper/2507.10535"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2412.08905","last_updated":"2024-12-12T03:37:41Z","snapshot_observed_at":"2026-08-05T04:04:21.846023Z","submitted_at":"2024-12-12T03:37:41Z","title":"Phi-4 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.08905","snapshot_observed_at":"2026-08-06T17:32:29.662597Z","title":"Phi-4 technical report","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:29.662597Z"},"links":{"cited_paper":"/paper/2412.08905","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:a1718b235e39d82ee91ad7bb8d607bf47d64680ccc04038a78cccdaef3f62ea5","observation_id":"0c9225aa-51d4-4752-85d4-1bcb44500da7","resolution":{"observed_at":"2026-08-06T17:32:29.662597Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-06T17:32:29.722527Z","title":"Gpt-4 technical report","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:29.722527Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:1845a8e25a431c5f8bc5a4713ad3ecfc60822d4da137b57277b048c745ecd1a2","observation_id":"9c8434c2-b472-434c-b1f1-b5318c68cdd3","resolution":{"observed_at":"2026-08-06T17:32:29.722527Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:42.676748Z","title":"Automated unit test improvement using large language models at meta","venue":null,"work_id":"aa33e1dd-bab4-442c-9fb3-d0a1866555bc","year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:29.910326Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:b3016f53a670aa0b9cbe00ef981d613d20fd1c05446da80cd77df947842ab401","observation_id":"e1e60d71-f17a-48b4-a3ee-b24feed19bc5","resolution":{"observed_at":"2026-08-06T17:32:42.690068Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:42.489668Z","title":"Claude 3.7","venue":null,"work_id":"4014125b-01cd-4341-ade8-195666915441","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:30.002094Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:6b6523719dcfdf7e4eba07e1d37d1ad8323df7c57afcee23c2c861401ea0f88e","observation_id":"62df6330-63b9-4093-95d4-df8bb968ef8c","resolution":{"observed_at":"2026-08-06T17:32:42.581478Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:42.290037Z","title":"Claude 4","venue":null,"work_id":"0c2d82ad-6031-4de2-a40d-c63f141ac9b1","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:30.072314Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:493970585ecc206c48060bd7c43b9395a7a68c8e3091de3d3d2dd614f08d370c","observation_id":"9341a860-3a15-4d0d-b84c-f39b31c1ff6a","resolution":{"observed_at":"2026-08-06T17:32:42.378283Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2108.07732","last_updated":"2021-08-16T03:57:30Z","snapshot_observed_at":"2026-08-02T19:23:53.535075Z","submitted_at":"2021-08-16T03:57:30Z","title":"Program Synthesis with Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.07732","snapshot_observed_at":"2026-08-06T17:32:30.162463Z","title":"Program synthesis with large language models","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:30.162463Z"},"links":{"cited_paper":"/paper/2108.07732","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:3b4e40687c796c6582337ce7ca80a7f592cca96a7bfa862afd65ce96f48fcf5f","observation_id":"3bd8cac9-39a9-4a64-95fb-fc03c8fa2508","resolution":{"observed_at":"2026-08-06T17:32:30.162463Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:30.256357Z","title":"Codet: Code generation with generated tests","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:30.256357Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:a20e5bcd1caa1959b7901c6b27bd8515b2aaaf655fe19862405e069df58c8b39","observation_id":"0e4ec585-5f9d-41e7-979f-cb0bb4d58296","resolution":{"observed_at":"2026-08-06T17:32:30.256357Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2107.03374","last_updated":"2021-07-14T17:16:02Z","snapshot_observed_at":"2026-08-08T11:58:24.516369Z","submitted_at":"2021-07-07T17:41:24Z","title":"Evaluating Large Language Models Trained on Code","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2107.03374","snapshot_observed_at":"2026-08-06T17:32:30.358202Z","title":"Evaluating large language models trained on code","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:30.358202Z"},"links":{"cited_paper":"/paper/2107.03374","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:e525455a014a0fefc5a1ab23b37c39fbafaa91e3ca95dfd2932885686835c9d2","observation_id":"45ae6b93-0aed-4333-8e5a-2843f38df120","resolution":{"observed_at":"2026-08-06T17:32:30.358202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:42.158114Z","title":"Teaching large language models to self-debug","venue":null,"work_id":"196967b6-0711-4769-9583-9027c1e19496","year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:30.467485Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:c874b0d8118fb11c339800399e09e93434144115359bfb4ec80388594c644a10","observation_id":"d409d1fb-c907-4c60-987b-00016a2eb356","resolution":{"observed_at":"2026-08-06T17:32:42.179788Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:30.562154Z","title":"Rm-r1: Reward modeling as reasoning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:30.562154Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:66641a23b92543d89a28e9cf88b51be743142a50a00f1a2850b75204e0637d37","observation_id":"8f725fbe-d2e9-48e0-9fb4-00a9eb87b637","resolution":{"observed_at":"2026-08-06T17:32:30.562154Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.16400","last_updated":"2025-06-05T17:59:12Z","snapshot_observed_at":"2026-08-09T23:26:41.871607Z","submitted_at":"2025-05-22T08:50:47Z","title":"AceReason-Nemotron: Advancing Math and Code Reasoning through Reinforcement Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.16400","snapshot_observed_at":"2026-08-06T17:32:30.634578Z","title":"Acereason-nemotron: Advancing math and code reasoning through reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:30.634578Z"},"links":{"cited_paper":"/paper/2505.16400","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:c20ca80e22924b6b98b105b961c9ff2defad66b59665a2eeabc1bd4b5e67f8b0","observation_id":"6712258d-8282-40f6-ad1f-31d0bc269b16","resolution":{"observed_at":"2026-08-06T17:32:30.634578Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:41.985542Z","title":null,"venue":null,"work_id":"69a6f000-ad38-4cf0-8c2a-da297d44ec86","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:30.697316Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:48839efff979458062cc8f1768c5209809a65ee570c7e1e11f7a5aa6840786c8","observation_id":"695a6cdc-7378-42b7-bcb5-426c16d294c5","resolution":{"observed_at":"2026-08-06T17:32:42.080154Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-06T17:32:30.805802Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:30.805802Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:fd8ef3eabb0a25dbb9305df0e63c369efcef8956349aa2b791a42a2be5ab0d14","observation_id":"410a27dd-5851-4e7c-8697-13ed5db91dc1","resolution":{"observed_at":"2026-08-06T17:32:30.805802Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.06782","last_updated":"2024-06-02T16:16:49Z","snapshot_observed_at":"2026-08-07T16:46:33.307475Z","submitted_at":"2023-08-13T14:35:50Z","title":"PentestGPT: An LLM-empowered Automatic Penetration Testing Tool","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.06782","snapshot_observed_at":"2026-08-06T17:32:30.900133Z","title":"Pentestgpt: An llm-empowered automatic penetration testing tool","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:30.900133Z"},"links":{"cited_paper":"/paper/2308.06782","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:c988ce9bc51195ebca54a623b6b7bdab0d3e86438fdb7eacd5b70bc74bfc6bfc","observation_id":"3b843fee-cf32-4efb-88a2-681611e71858","resolution":{"observed_at":"2026-08-06T17:32:30.900133Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14723","last_updated":"2025-02-03T18:57:05Z","snapshot_observed_at":"2026-08-10T14:49:22.326836Z","submitted_at":"2025-01-24T18:58:40Z","title":"CodeMonkeys: Scaling Test-Time Compute for Software Engineering","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.14723","snapshot_observed_at":"2026-08-06T17:32:30.995056Z","title":"Codemonkeys: Scaling test-time compute for software engineering","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:30.995056Z"},"links":{"cited_paper":"/paper/2501.14723","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:b64a3f0e383280cf27f964b1c30dbce05b0cb9875a5a6997c9fddc3f0ba64faf","observation_id":"a506cdaf-03d9-4bbf-8546-4787acd178c7","resolution":{"observed_at":"2026-08-06T17:32:30.995056Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13820","last_updated":"2025-07-30T14:58:42Z","snapshot_observed_at":"2026-08-07T18:04:41.000672Z","submitted_at":"2025-02-19T15:32:11Z","title":"Scoring Verifiers: Evaluating Synthetic Verification for Code and Reasoning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13820","snapshot_observed_at":"2026-08-06T17:32:31.064570Z","title":"Scoring verifiers: Eval- uating synthetic verification for code and reasoning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:31.064570Z"},"links":{"cited_paper":"/paper/2502.13820","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:d7df7e0fe194d80834f7082ed22badd1b33b751deb997439f0140e7e278caf82","observation_id":"e5c1bca7-a3d5-4dbe-8616-345224c88e81","resolution":{"observed_at":"2026-08-06T17:32:31.064570Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:41.759917Z","title":"Gonzalez, and Ion Stoica","venue":null,"work_id":"b963f2e9-ed30-4b5e-a6ba-5204cc6810fe","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:31.155235Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:2b40e6b6bc576cee221e5898dcc5dd39a1fc6a9dd1df6625e107960812e85eb4","observation_id":"95dd83da-1adb-44f4-ad80-4d83218f093d","resolution":{"observed_at":"2026-08-06T17:32:41.850357Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-06T17:32:31.253381Z","title":"The llama 3 herd of models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:31.253381Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:8fdae52c6da814e0ac9b9b898672d1974af5bbb1ccfdaa4dd9ea8ec294b1b2e6","observation_id":"d941a680-1c4d-4da4-924e-50dd858da513","resolution":{"observed_at":"2026-08-06T17:32:31.253381Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15594","last_updated":"2025-10-19T10:32:43Z","snapshot_observed_at":"2026-08-02T10:23:50.881300Z","submitted_at":"2024-11-23T16:03:35Z","title":"A Survey on LLM-as-a-Judge","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15594","snapshot_observed_at":"2026-08-06T17:32:31.353188Z","title":"A survey on llm-as-a-judge","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:31.353188Z"},"links":{"cited_paper":"/paper/2411.15594","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:6d70704b3d77c2a0422e4ec3a34a0536661fd151b2d61f3ac2bc4d5c05440929","observation_id":"316e3584-dd13-49d2-bfa7-fb34d882d5d8","resolution":{"observed_at":"2026-08-06T17:32:31.353188Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.02246","last_updated":"2025-03-04T03:48:23Z","snapshot_observed_at":"2026-08-07T17:31:39.128866Z","submitted_at":"2025-03-04T03:48:23Z","title":"From Code to Courtroom: LLMs as the New Software Judges","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.02246","snapshot_observed_at":"2026-08-06T17:32:31.444375Z","title":"From code to courtroom: Llms as the new software judges","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:31.444375Z"},"links":{"cited_paper":"/paper/2503.02246","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:6ec9a04e9a9917da83a576426dae5485d52a69f3aa26c9d849b5ff2ef5a5c8fb","observation_id":"6c19e092-2e61-4550-8117-b1a7ec4a4f62","resolution":{"observed_at":"2026-08-06T17:32:31.444375Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:41.591732Z","title":"An empirical study on fine-tuning large language models of code for automated program repair","venue":null,"work_id":"81fe7dc1-7695-444c-a920-c74f1d0819d7","year":2023},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:31.535547Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:eddc56804d50e295ff87b78701df9e0824560d925f26c0177551138d77a36eef","observation_id":"3647b23d-a7eb-49a6-befb-0dbbc17e4742","resolution":{"observed_at":"2026-08-06T17:32:41.666376Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:41.243654Z","title":"Livecodebench: Holistic and contamination free evaluation of large language models for code","venue":null,"work_id":"3a20a727-33d5-4194-b5c6-924ac2dea58a","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:31.724734Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:0b27da131c1e97313b0625dc2e8ff5f4bee76e4cbdc53d1e4b6e446ba3b7914e","observation_id":"eaa332d2-4a91-4484-ba4b-004a39388cfc","resolution":{"observed_at":"2026-08-06T17:32:41.298959Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:41.067413Z","title":"Self-planning code generation with large language models","venue":null,"work_id":"4f043466-0623-449a-9c8f-7154a68b5c06","year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:31.808480Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:3d23947427c471aea323572f69b7d782c8c636244c2d827a0504b467e494c2f5","observation_id":"578fa9a7-ae6c-47e5-849d-3318038fe9c7","resolution":{"observed_at":"2026-08-06T17:32:41.116337Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:40.870737Z","title":"Critiquellm: Towards an informative critique generation model for evaluation of large language model generation","venue":null,"work_id":"1ec4df92-a32f-4d41-8c3f-e48f81b7252c","year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:31.884269Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:b148a2d4bba1c5fadeab238a9482065926a74185fbd894118e27146534769c6c","observation_id":"ef81ede8-6741-4d9a-be3c-9706e88a8122","resolution":{"observed_at":"2026-08-06T17:32:40.948898Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:40.643469Z","title":"Welleck, Graham Neubig, Moontae Lee, Kyungjae Lee, and Minjoon Seo","venue":null,"work_id":"239bf8f2-ca26-4f06-b9a7-d54f95e59221","year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:31.947036Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:51f769aa30fd18b0288c8176a097f82cea53bbe2c7212bba48b3ff0fd0bb93f2","observation_id":"a3d91791-856c-4d30-bea3-b492ecddfe3b","resolution":{"observed_at":"2026-08-06T17:32:40.717471Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:40.531484Z","title":"Overfitting in semantics-based automated program repair","venue":null,"work_id":"d1c444dd-8b90-4e14-a16c-45bf476e32eb","year":2018},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:31.986536Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:94d197447d17be167cb1854820cd4f38dbb25ae1a46a295cd486357f4af04488","observation_id":"b74c1185-6ad2-44c5-af6c-4095df969310","resolution":{"observed_at":"2026-08-06T17:32:40.627958Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:40.309690Z","title":"Generative judge for evaluating alignment","venue":null,"work_id":"40278fa4-b41e-42c6-b573-bf4bfb5fb373","year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:32.050286Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:b455187b4f32f36fa7e4e1d4929ca642353fd21264daafb73f08e7431f3781c3","observation_id":"2ddd0d21-3a93-4d42-b139-cdbb69f0486d","resolution":{"observed_at":"2026-08-06T17:32:40.377848Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:40.121493Z","title":"Competition-level code generation with alphacode","venue":null,"work_id":"5cce0596-47d9-42cb-b554-b3e5ac88aa7b","year":2022},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:32.125628Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:3249c0e34b6273bf80d212f196a43b8ab5219dfecedfa594693dc8d7d55ef3f5","observation_id":"73900c1d-eef2-4ca9-9e17-3bbe1bb4b175","resolution":{"observed_at":"2026-08-06T17:32:40.177026Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:39.911338Z","title":"Llms for relational reasoning: How far are we? In Proceedings of the 1st International Workshop on Large Language Models for Code, pages 119–126, 2024","venue":null,"work_id":"4fa89a2d-a66e-4def-899b-c63ee82fff16","year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:32.215870Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:cbbaef47f066bab7b70ed2ecaf0280584530cbc427dde651d6b244db76fe830a","observation_id":"e923a080-abd6-415f-a6fd-116562346624","resolution":{"observed_at":"2026-08-06T17:32:39.982049Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:39.703264Z","title":"RM-bench: Benchmarking reward models of language models with subtlety and style","venue":null,"work_id":"cc45c4d8-76c7-45ab-8320-5e31c0d0c31c","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:32.300774Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:6b2b640abc4ade41cdaf672aa3aaac3258fc4860ae5cd3396a8fcbccbea23b9e","observation_id":"82359399-10d2-4abb-8e8b-46f46f96a630","resolution":{"observed_at":"2026-08-06T17:32:39.801044Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:39.434975Z","title":"Deepcoder: A fully open-source 14b coder at o3-mini level","venue":null,"work_id":"0848e680-27ab-4e16-a490-6639a313da12","year":null},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:32.397991Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:bc44a541a621b6cf6ad48e7009d0e278151502ba9dbfef3218e8cd008107aca4","observation_id":"631f1674-bce1-43d5-83b6-4f0f63460f6d","resolution":{"observed_at":"2026-08-06T17:32:39.578462Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00215","last_updated":"2024-06-28T19:53:17Z","snapshot_observed_at":"2026-08-10T12:31:19.058935Z","submitted_at":"2024-06-28T19:53:17Z","title":"LLM Critics Help Catch LLM Bugs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.00215","snapshot_observed_at":"2026-08-06T17:32:32.445876Z","title":"Llm critics help catch llm bugs","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:32.445876Z"},"links":{"cited_paper":"/paper/2407.00215","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:0bf412a7c69f9a83e6d9fc7004461db0da050a3a38b93e3e18e75ccb51e2d40a","observation_id":"6868fb9f-f436-4f7d-959a-d7d92e1ba5c1","resolution":{"observed_at":"2026-08-06T17:32:32.445876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:39.251400Z","title":"Swt-bench: Testing and validating real- world bug-fixes with code agents","venue":null,"work_id":"a5f1d2bd-1108-4bb8-a111-104ec21d89ca","year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:32.554314Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:f9bc532961fa6d000ccf6fb5940346af328034690c55fa882bca331ff47712fc","observation_id":"829352c5-1163-4e92-8643-563012c9b409","resolution":{"observed_at":"2026-08-06T17:32:39.338250Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:39.066987Z","title":"Wainwright, Pamela Mishkin, Chong Zhang, Sandhini Agarwal, Katarina Slama, Alex Ray, John Schulman, Jacob Hilton, Fraser Kelton, Luke E","venue":null,"work_id":"cd754242-3fee-4abb-b7d0-8bf7613a7c55","year":2022},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:32.619413Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:5eb76060bae6bef9a6e28571973b256dedb8246a5269edb79bb613a01516d96c","observation_id":"472ed59f-d8a6-4962-bb05-5172541db030","resolution":{"observed_at":"2026-08-06T17:32:39.158021Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:32.748374Z","title":"M-prometheus: A suite of open multilingual llm judges","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:32.748374Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:b6918be29be6af4172143a780fac06d22be9b98c321612acca3f139d440d8eb7","observation_id":"5ebb1ec5-b98a-4307-8344-71d3e751d1ce","resolution":{"observed_at":"2026-08-06T17:32:32.748374Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.10297","last_updated":"2020-09-27T04:07:11Z","snapshot_observed_at":"2026-08-01T07:33:26.380394Z","submitted_at":"2020-09-22T03:10:49Z","title":"CodeBLEU: a Method for Automatic Evaluation of Code Synthesis","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.10297","snapshot_observed_at":"2026-08-06T17:32:32.851096Z","title":"Codebleu: a method for automatic evaluation of code synthesis","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:32.851096Z"},"links":{"cited_paper":"/paper/2009.10297","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:944baa40ef664b613ac5fe0862394ca7e7efa2251a959a9abfb460453cd2664b","observation_id":"37fc0660-7375-480d-81b2-a5f9e0a455bc","resolution":{"observed_at":"2026-08-06T17:32:32.851096Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:38.700270Z","title":"Skywork critic model series","venue":null,"work_id":"066f62eb-2e82-44f0-ba80-7358ae83470c","year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:32.948957Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:bdebc9a0899ce25c70c9a70a6237c86f0a23cc0a69fc1c3ffdc24401c3727ec0","observation_id":"021bad87-7e6c-4c79-9237-9370d53118ce","resolution":{"observed_at":"2026-08-06T17:32:38.843015Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03314","last_updated":"2024-08-06T17:35:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:35:05Z","title":"Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03314","snapshot_observed_at":"2026-08-06T17:32:33.016659Z","title":"Scaling llm test-time compute optimally can be more effective than scaling model parameters","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.016659Z"},"links":{"cited_paper":"/paper/2408.03314","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:ef913fd53a4484034231fa9801fc6965860107fd0c1d1ff5a7f1b0baa2625e80","observation_id":"0501d0ec-bd1a-40fc-9468-b84909a2b173","resolution":{"observed_at":"2026-08-06T17:32:33.016659Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:38.543059Z","title":"Judgebench: A benchmark for evaluating LLM-based judges","venue":null,"work_id":"3cf8cb53-f1d3-4471-90db-72a01faf713c","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.074682Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:1d869c9a12655a48277d43a94c1fa9b854015faddf994975ad6c67b4dfcd9ecd","observation_id":"8511f9d0-f646-4aae-9251-be4967b4aa68","resolution":{"observed_at":"2026-08-06T17:32:38.624158Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:38.359660Z","title":"Code repair with LLMs gives an exploration-exploitation tradeoff","venue":null,"work_id":"f66f49ec-7ac0-4316-8440-047f960d6103","year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.171459Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:c780ae3a62575b7f653ecde7ea79d8939fbe90a28ec439ab066be9eefee51924","observation_id":"c7f9eda7-12be-4143-a706-61e625c8f76d","resolution":{"observed_at":"2026-08-06T17:32:38.439815Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:33.228386Z","title":"Qwq-32b: Embracing the power of reinforcement learning, March 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.228386Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:7e6dd64168e5629c6f18355aa052da62faa6940dbab76927aa0e9f1408ad10d0","observation_id":"c5763663-0db8-4bcc-9788-2daf9c7148fb","resolution":{"observed_at":"2026-08-06T17:32:33.228386Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:38.043255Z","title":"Can llms replace human evaluators? an empirical study of llm-as-a-judge in software engineering","venue":null,"work_id":"f2de2f64-a0ba-4e85-aa6a-884189e4d751","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.302701Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:d09e6ef3cc9639305d4743b4519bab7bbd389249518819c48999f1e63733673c","observation_id":"140f308e-9e28-4b3a-b437-16554802428b","resolution":{"observed_at":"2026-08-06T17:32:38.150125Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.02666","last_updated":"2024-08-08T17:09:58Z","snapshot_observed_at":"2026-08-08T10:45:45.471332Z","submitted_at":"2024-08-05T17:57:02Z","title":"Self-Taught Evaluators","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.02666","snapshot_observed_at":"2026-08-06T17:32:33.368009Z","title":"Self-taught evaluators","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.368009Z"},"links":{"cited_paper":"/paper/2408.02666","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:89713b8354f3115b947b584029fb5eea599842ed0e7b4ea6e57031b0e9519c6c","observation_id":"ac0be8b9-04c1-4ea2-a14f-64a21bb03f77","resolution":{"observed_at":"2026-08-06T17:32:33.368009Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:37.873991Z","title":"PandaLM: An automatic evaluation benchmark for LLM instruction tuning optimization","venue":null,"work_id":"229d215f-e628-45fc-9c3e-6b62ed9ef769","year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.423579Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:f0e0b8baa62fc1660a1d21bcda3a72b87bca02e5f8607e47ad38199c15474444","observation_id":"96dfe862-aef5-47d9-8556-c1c967cc724d","resolution":{"observed_at":"2026-08-06T17:32:37.938874Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:37.630252Z","title":"Chi, Tatsunori Hashimoto, O","venue":null,"work_id":"7f1c7d3e-0041-4cf3-a23e-d1a76fe37a9a","year":2022},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.480092Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:efd807b42e12385ba91acd10cb3c436b365d4593e41b60455e6039fe6cc5ca25","observation_id":"d02dbe15-a1e3-44b9-8637-b54352e5b1d8","resolution":{"observed_at":"2026-08-06T17:32:37.753167Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:37.341324Z","title":"Weyssow, Aton Kamanda, Xin Zhou, and H","venue":null,"work_id":"e348e673-976e-4eb9-b23c-6fa989346666","year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.542361Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:6452b57e4ff216e212087abc90752f80f562e504fa0f5e5a9e5dd088cb244a65","observation_id":"88c200d2-e87a-4020-99f4-0866db8a631e","resolution":{"observed_at":"2026-08-06T17:32:37.467984Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.17564","last_updated":"2023-12-21T06:21:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-30T17:30:36Z","title":"BloombergGPT: A Large Language Model for Finance","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.17564","snapshot_observed_at":"2026-08-06T17:32:33.592955Z","title":"Bloomberggpt: A large language model for finance","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.592955Z"},"links":{"cited_paper":"/paper/2303.17564","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:472370cb605fbdb4444749b2cff0cec5703ab95520d683bfdc0320693356e07b","observation_id":"315537e2-45d4-4d6b-8b39-5b45057ab3ba","resolution":{"observed_at":"2026-08-06T17:32:33.592955Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-08-06T17:32:33.649376Z","title":"Qwen3 technical report","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.649376Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:781fb7f23cb68e621cc5d2c61ea4ddf3d562f6fb9dba878c84e54cb9327ec816","observation_id":"e1b57fa7-d8c3-4db2-9623-1166c2b32fa1","resolution":{"observed_at":"2026-08-06T17:32:33.649376Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-06T17:32:33.704634Z","title":"Qwen2.5 technical report","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.704634Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:11b11964412f44e1e4ba548656c8187f330bfb2349e4101ec41648615234222d","observation_id":"c8630a18-a370-41d3-b810-fce5a54e322b","resolution":{"observed_at":"2026-08-06T17:32:33.704634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19502","last_updated":"2025-05-26T04:29:14Z","snapshot_observed_at":"2026-08-08T05:31:17.095413Z","submitted_at":"2025-05-26T04:29:14Z","title":"CODE-DITING: A Reasoning-Based Metric for Functional Alignment in Code Evaluation","version":1},"cited_work":{"arxiv_id":"2505.19502","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.19502","snapshot_observed_at":"2026-08-06T17:32:34.591005Z","title":"CODE-DITING: A Reasoning-Based Metric for Functional Alignment in Code Evaluation","venue":"cs.SE","work_id":"761a9802-1014-408b-ae6d-a7bd93131e61","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.789468Z"},"links":{"cited_paper":"/paper/2505.19502","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:6049fb8779f69afee240b81b0e3e6bf648152bed9569b6fe9a289ed1f86ec4c4","observation_id":"3578691e-6982-4de8-8a26-b88ee4ce7fbf","resolution":{"observed_at":"2026-08-06T17:32:34.645100Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:33.841936Z","title":"Fingpt: Open-source financial large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.841936Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:b59ce6812470585bbca72cebdc5141243f1c8ac48a57036805a37999b1153487","observation_id":"03322681-7dee-498c-a3c3-3894d5dd2a95","resolution":{"observed_at":"2026-08-06T17:32:33.841936Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:37.111759Z","title":"Demystifying long chain-of- thought reasoning in llms, 2025","venue":null,"work_id":"db0d7318-96f4-4fe6-aff7-02b55ae1fc7c","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.881950Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:693bfeda78db6a39103d7a2cc46991e1358e40b9e435ac29c5756e7a91857bb8","observation_id":"c3182861-bd31-4aff-8784-a6e4d46520d5","resolution":{"observed_at":"2026-08-06T17:32:37.208972Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01718","last_updated":"2025-05-24T04:36:48Z","snapshot_observed_at":"2026-08-10T05:42:26.340829Z","submitted_at":"2025-02-03T18:46:04Z","title":"ACECODER: Acing Coder RL via Automated Test-Case Synthesis","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01718","snapshot_observed_at":"2026-08-06T17:32:33.941396Z","title":"Acecoder: Acing coder rl via automated test-case synthesis","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.941396Z"},"links":{"cited_paper":"/paper/2502.01718","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:063c000cc3cba4a8975343529b8b872355407e10d79f60208a6692a4aeb3c3c3","observation_id":"de824f34-39f5-4063-9f92-02c42a138841","resolution":{"observed_at":"2026-08-06T17:32:33.941396Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:36.818079Z","title":"Codecriticbench: A holistic code critique benchmark for large language models, 2025","venue":null,"work_id":"28831811-ff04-4c8e-80f9-bda739e00b3c","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.991009Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:56c8ef37fea749341575f39035dcff2f1ed7bf3ab71f44902e1c6ac8e949d562","observation_id":"14f188bb-f344-4d63-b28a-ffffff9a306f","resolution":{"observed_at":"2026-08-06T17:32:36.936289Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:36.528012Z","title":null,"venue":null,"work_id":"6af7d204-bcbe-49e9-9875-6e4c941889e8","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:34.024035Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:d0ab191a7a77476e3c6fda910a8fcab9d21743c09db86e768dab3c28316897b0","observation_id":"e11e4bc7-896e-4b54-b34e-7ea4f720d65c","resolution":{"observed_at":"2026-08-06T17:32:36.639897Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:34.062676Z","title":"Judging llm-as-a-judge with mt-bench and chatbot arena","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:34.062676Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:32ca01a32eea90d20febd9d60595aa868062de8ea5d58afbd1aab7a7630e4404","observation_id":"9f89e6cd-5739-484b-be82-2f5872ce34bf","resolution":{"observed_at":"2026-08-06T17:32:34.062676Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:36.241147Z","title":"RMB: Compre- hensively benchmarking reward models in LLM alignment","venue":null,"work_id":"5353d0c8-2cd0-489e-abe8-95947fd0ad23","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:34.116865Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:de805a851a099c5db47c0434608f13a7f66f36bc1734102505d7134ff88cce01","observation_id":"7fd513e5-0d7e-4684-89bc-9a3e21ce592c","resolution":{"observed_at":"2026-08-06T17:32:36.342972Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:35.886198Z","title":"Leveraging large language model for automatic patch correctness assessment","venue":null,"work_id":"7381dd86-441e-4f03-bf46-2e8f8be9dbd5","year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:34.165260Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:1cdf3a41243f8a161d89270e8b3a44f062ecee8fd2e88e65ec6bf0529bb49e73","observation_id":"748fb1ef-1c7e-4a3a-a8da-615210c80fcb","resolution":{"observed_at":"2026-08-06T17:32:36.057972Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:35.582317Z","title":"Evaluating judges as evaluators: The JETTS benchmark of LLM-as-judges as test-time scaling evaluators","venue":null,"work_id":"d2f054ef-e77f-4a0b-9640-400356c07854","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:34.238243Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:47d297130e7a070757c63a34ee289f77b3c6666b77ca29a26592a3340171e6bc","observation_id":"31548f35-18a2-4c74-b9fa-d3c5a84bfa18","resolution":{"observed_at":"2026-08-06T17:32:35.747660Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:35.314175Z","title":"JudgeLM: Fine-tuned large language models are scalable judges","venue":null,"work_id":"746bb14e-f04e-4bde-8ad2-9dfce059374f","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:34.292050Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:46350c696d355310f6299c2059973a35a1f012f9a0e6121e327e7fd2b0f3afcf","observation_id":"a1aa5af6-0d44-43b3-ba26-2d6a0bcf856e","resolution":{"observed_at":"2026-08-06T17:32:35.431422Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:41.383474Z","title":null,"venue":null,"work_id":"2862e88a-99db-465c-91e1-128a1691f13d","year":2023},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":1174,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:31.633118Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:44e575777dccdcca73e48092d192f9c18a942feeefdd747b41266a51d9bc60f3","observation_id":"c34bd653-6094-41c6-abab-36d6b78352cc","resolution":{"observed_at":"2026-08-06T17:32:41.495630Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-09T16:40:01.179500Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks"},"reference_resolution":{"displayed":61,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":29,"verified_exact":1,"verified_fuzzy":31},"total_outbound_references":61},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 61 of 61 outbound references and 11 inbound Pith citation observations for arXiv:2507.10535."}