{"as_of":"2026-08-20T15:03:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d1e19deb5d002baaa0d4060cf7e16a4eeb52b9042a7ee564eb914eb958efd7fb","coverage":[{"denominator":23,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":23,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-17T19:57:06.015270Z","state":"measured"},{"denominator":25,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":25,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-20T06:33:59.587034+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-03T17:03:47.442858Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2511.16858","last_updated":"2026-04-03T16:15:46Z","snapshot_observed_at":"2026-07-06T22:36:35.545474Z","submitted_at":"2025-11-20T23:55:56Z","title":"Investigating Test Overfitting on SWE-bench","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2511.16858","snapshot_observed_at":"2026-07-14T06:59:44.621968Z","title":"Investigating test overfitting on SWE-bench,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.11111","last_updated":"2026-07-13T05:41:01Z","snapshot_observed_at":"2026-08-14T22:48:49.923228Z","submitted_at":"2026-07-13T05:41:01Z","title":"Know Before Fix: QA-Driven Repository Knowledge Acquisition for Software Issue Resolution","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-07-14T06:59:44.621968Z"},"links":{"cited_paper":"/paper/2511.16858","citing_paper":"/paper/2607.11111"},"observation_digest":"sha256:db2bf497407cfbf2f33f1774cd4b1a845fb90b6432c4b1c9fd049df9297d2ee8","observation_id":"39aa7983-2c4a-4a50-8930-a2c75f031ac8","resolution":{"observed_at":"2026-07-14T06:59:44.621968Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2511.16858","last_updated":"2026-04-03T16:15:46Z","snapshot_observed_at":"2026-07-06T22:36:35.545474Z","submitted_at":"2025-11-20T23:55:56Z","title":"Investigating Test Overfitting on SWE-bench","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2511.16858","snapshot_observed_at":"2026-08-03T17:03:47.442858Z","title":"Is the cure still worse than the disease? test overfitting by llms in automated program repair,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.28928","last_updated":"2026-07-31T01:04:24Z","snapshot_observed_at":"2026-08-17T12:32:14.321382Z","submitted_at":"2026-07-31T01:04:24Z","title":"Automated Testing and Repair for Verified Compilers Generated by a Coding Agent","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-03T17:03:47.442858Z"},"links":{"cited_paper":"/paper/2511.16858","citing_paper":"/paper/2607.28928"},"observation_digest":"sha256:4a79ed2919eaaf9e9edc1ecbb4ab2132715d11f8088b3df3725f627f65511881","observation_id":"d5eb9daa-19f0-41e1-8ea6-02c1dc52a707","resolution":{"observed_at":"2026-08-03T17:03:47.442858Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2511.16858/citation-record","integrity":"/paper/2511.16858/integrity","json":"/paper/2511.16858/citation-record.json","paper":"/paper/2511.16858"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2508.06365","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"19fbcd9e-fcd4-4504-90fc-1802e0a104ed","year":2025},"citing_paper":{"arxiv_id":"2511.16858","last_updated":"2026-04-03T16:15:46Z","snapshot_observed_at":"2026-07-06T22:36:35.545474Z","submitted_at":"2025-11-20T23:55:56Z","title":"Investigating Test Overfitting on SWE-bench","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-17T19:57:06.015270Z"},"links":{"citing_paper":"/paper/2511.16858"},"observation_digest":"sha256:9913f4054895046ae5a4407a013a7cc4fb4bc9d0f801ac452a387518110f2a92","observation_id":"3b508382-d91d-48cf-ae22-92843a0a0d3d","resolution":{"observed_at":"2026-05-17T20:00:10.782947Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02883","last_updated":"2024-12-03T22:38:05Z","snapshot_observed_at":"2026-08-16T07:01:24.337018Z","submitted_at":"2024-12-03T22:38:05Z","title":"TDD-Bench Verified: Can LLMs Generate Tests for Issues Before They Get Resolved?","version":1},"cited_work":{"arxiv_id":"2412.02883","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.02883","snapshot_observed_at":"2026-07-05T09:10:52.106373Z","title":"TDD-Bench verified: Can LLMs generate tests for issues before they get resolved?arXiv preprint arXiv:2412.02883","venue":null,"work_id":"b0a18d84-bbf2-47a5-9c44-5e3812f07b46","year":2024},"citing_paper":{"arxiv_id":"2511.16858","last_updated":"2026-04-03T16:15:46Z","snapshot_observed_at":"2026-07-06T22:36:35.545474Z","submitted_at":"2025-11-20T23:55:56Z","title":"Investigating Test Overfitting on SWE-bench","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-17T19:57:06.015270Z"},"links":{"cited_paper":"/paper/2412.02883","citing_paper":"/paper/2511.16858"},"observation_digest":"sha256:3429166344c4c776b1a456045a6fc3a1c3fed8510e5e4ff853ed4239778b1d74","observation_id":"2db7f73a-6e11-468d-b6f2-b729e96e90ae","resolution":{"observed_at":"2026-05-17T20:00:10.761621Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11638","last_updated":"2024-06-17T15:19:51Z","snapshot_observed_at":"2026-08-18T18:37:22.824874Z","submitted_at":"2024-06-17T15:19:51Z","title":"MASAI: Modular Architecture for Software-engineering AI Agents","version":1},"cited_work":{"arxiv_id":"2406.11638","doi":"10.48550/arxiv.2406.11638","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.11638","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Masai: Modular architecture for software-engineering ai agents","venue":"arXiv (Cornell University)","work_id":"18e446d2-0b0a-4946-9083-2e4547fcdfac","year":2024},"citing_paper":{"arxiv_id":"2511.16858","last_updated":"2026-04-03T16:15:46Z","snapshot_observed_at":"2026-07-06T22:36:35.545474Z","submitted_at":"2025-11-20T23:55:56Z","title":"Investigating Test Overfitting on SWE-bench","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-17T19:57:06.015270Z"},"links":{"cited_paper":"/paper/2406.11638","citing_paper":"/paper/2511.16858"},"observation_digest":"sha256:28db54b3b584d82b77772e3fb4ae1ccdd3242cff78fad59f281b68a07f661864","observation_id":"5825aa92-aecc-452e-a5dc-a24e71cd1151","resolution":{"observed_at":"2026-05-17T20:00:10.766675Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.11926","last_updated":"2025-03-14T23:50:34Z","snapshot_observed_at":"2026-08-17T15:42:03.133713Z","submitted_at":"2025-03-14T23:50:34Z","title":"Monitoring Reasoning Models for Misbehavior and the Risks of Promoting Obfuscation","version":1},"cited_work":{"arxiv_id":"2503.11926","doi":"10.48550/arxiv.2503.11926","metadata_source":"pith","pith_arxiv_id":"2503.11926","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Monitoring Reasoning Models for Misbehavior and the Risks of Promoting Obfuscation","venue":"cs.AI","work_id":"d428a378-91ee-4965-ae16-b2fd3977bbf2","year":2025},"citing_paper":{"arxiv_id":"2511.16858","last_updated":"2026-04-03T16:15:46Z","snapshot_observed_at":"2026-07-06T22:36:35.545474Z","submitted_at":"2025-11-20T23:55:56Z","title":"Investigating Test Overfitting on SWE-bench","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-17T19:57:06.015270Z"},"links":{"cited_paper":"/paper/2503.11926","citing_paper":"/paper/2511.16858"},"observation_digest":"sha256:de08edb92264b65cab513c081de42459ee989a561ed5b71290b8fcdf1035f485","observation_id":"c7ee6d44-dadb-4784-b2e1-0177e4f532c2","resolution":{"observed_at":"2026-05-21T07:24:13.179886Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"ad585be1-9a6c-4096-a440-c27ac89245a1","year":2023},"citing_paper":{"arxiv_id":"2511.16858","last_updated":"2026-04-03T16:15:46Z","snapshot_observed_at":"2026-07-06T22:36:35.545474Z","submitted_at":"2025-11-20T23:55:56Z","title":"Investigating Test Overfitting on SWE-bench","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-17T19:57:06.015270Z"},"links":{"citing_paper":"/paper/2511.16858"},"observation_digest":"sha256:47dd242ad82fa8403eec3eb4b4eff34af2406fe07e7831a6e61d9d416acc1674","observation_id":"ab4433ae-abd7-4c07-a8a1-2968903a52df","resolution":{"observed_at":"2026-05-17T20:00:11.352875Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"68ada17c-c18b-4131-8a23-86d27bb52845","year":2024},"citing_paper":{"arxiv_id":"2511.16858","last_updated":"2026-04-03T16:15:46Z","snapshot_observed_at":"2026-07-06T22:36:35.545474Z","submitted_at":"2025-11-20T23:55:56Z","title":"Investigating Test Overfitting on SWE-bench","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-17T19:57:06.015270Z"},"links":{"citing_paper":"/paper/2511.16858"},"observation_digest":"sha256:4e24e5536c85eba4cea19ca29d9dcf835f08ce9fc46937549eb3b906168d8c61","observation_id":"457b3b2a-8ed2-483f-a3be-0c2d89db946e","resolution":{"observed_at":"2026-05-17T20:00:11.347606Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Jimenez, John Yang, Kevin Liu, and Aleksander Madry","venue":null,"work_id":"8c3d1a18-9a37-4246-ac7b-496dc2840ad2","year":2024},"citing_paper":{"arxiv_id":"2511.16858","last_updated":"2026-04-03T16:15:46Z","snapshot_observed_at":"2026-07-06T22:36:35.545474Z","submitted_at":"2025-11-20T23:55:56Z","title":"Investigating Test Overfitting on SWE-bench","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-17T19:57:06.015270Z"},"links":{"citing_paper":"/paper/2511.16858"},"observation_digest":"sha256:0a753f6b1f003c53c4236289678b582806acefbfae33a71323280dbc7434110a","observation_id":"f0af13e9-53dc-4ba2-8189-3d86dcd7c6c9","resolution":{"observed_at":"2026-05-17T20:00:11.345076Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14723","last_updated":"2025-02-03T18:57:05Z","snapshot_observed_at":"2026-08-18T08:06:30.102423Z","submitted_at":"2025-01-24T18:58:40Z","title":"CodeMonkeys: Scaling Test-Time Compute for Software Engineering","version":2},"cited_work":{"arxiv_id":"2501.14723","doi":"10.48550/arxiv.2501.14723","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.14723","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2501.14723 , year=","venue":"ArXiv.org","work_id":"6d27ff89-d938-4164-bffa-112a80328b14","year":2025},"citing_paper":{"arxiv_id":"2511.16858","last_updated":"2026-04-03T16:15:46Z","snapshot_observed_at":"2026-07-06T22:36:35.545474Z","submitted_at":"2025-11-20T23:55:56Z","title":"Investigating Test Overfitting on SWE-bench","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-17T19:57:06.015270Z"},"links":{"cited_paper":"/paper/2501.14723","citing_paper":"/paper/2511.16858"},"observation_digest":"sha256:d1ab3fcd37e2f0e5f2698c4a9d1016c99ea878a487a0055c2bfe3c1ebe9c0aa5","observation_id":"8f7f4a26-332f-4616-b31f-afa61ab3c5a9","resolution":{"observed_at":"2026-05-17T20:00:10.756853Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.23370","last_updated":"2025-07-31T09:37:22Z","snapshot_observed_at":"2026-08-17T14:31:14.268621Z","submitted_at":"2025-07-31T09:37:22Z","title":"Trae Agent: An LLM-based Agent for Software Engineering with Test-time Scaling","version":1},"cited_work":{"arxiv_id":"2507.23370","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.23370","snapshot_observed_at":"2026-07-04T08:19:43.951695Z","title":"Trae agent: An llm-based agent for software engineering with test-time scaling","venue":null,"work_id":"9bbaf3fb-3f46-415d-bc2c-ecf1cfdd0924","year":2025},"citing_paper":{"arxiv_id":"2511.16858","last_updated":"2026-04-03T16:15:46Z","snapshot_observed_at":"2026-07-06T22:36:35.545474Z","submitted_at":"2025-11-20T23:55:56Z","title":"Investigating Test Overfitting on SWE-bench","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-17T19:57:06.015270Z"},"links":{"cited_paper":"/paper/2507.23370","citing_paper":"/paper/2511.16858"},"observation_digest":"sha256:b80db4ca854cfb74e3a7e18444081670501fee75fe9cc7b22e8e142a548e97df","observation_id":"0d641755-7e23-4db2-970a-f0b3a7bc36d3","resolution":{"observed_at":"2026-05-17T20:00:10.807369Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.07164","last_updated":"2025-04-09T17:55:19Z","snapshot_observed_at":"2026-08-16T12:42:37.309899Z","submitted_at":"2025-04-09T17:55:19Z","title":"R2E-Gym: Procedural Environments and Hybrid Verifiers for Scaling Open-Weights SWE Agents","version":1},"cited_work":{"arxiv_id":"2504.07164","doi":"10.48550/arxiv.2504.07164","metadata_source":"pith","pith_arxiv_id":"2504.07164","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"R2E-gym: Procedural environments and hybrid verifiers for scaling open-weights SWE agents","venue":"cs.SE","work_id":"d6614c64-acca-4134-bd4e-80ec66ffdf31","year":2025},"citing_paper":{"arxiv_id":"2511.16858","last_updated":"2026-04-03T16:15:46Z","snapshot_observed_at":"2026-07-06T22:36:35.545474Z","submitted_at":"2025-11-20T23:55:56Z","title":"Investigating Test Overfitting on SWE-bench","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-17T19:57:06.015270Z"},"links":{"cited_paper":"/paper/2504.07164","citing_paper":"/paper/2511.16858"},"observation_digest":"sha256:ea989e649a0f56aa34e0795cc4b47f48b2c7e866ddb4f8b8431b196ce5d1d535","observation_id":"49b41f7a-1411-4e80-9d85-40255ac79a77","resolution":{"observed_at":"2026-05-17T20:00:10.787186Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Jimenez, John Yang, Alexander Wettig, Shunyu Yao, Kexin Pei, Ofir Press, and Karthik Narasimhan","venue":null,"work_id":"ef54dc4d-4af6-40fb-bcf0-a92baedbcb6e","year":2024},"citing_paper":{"arxiv_id":"2511.16858","last_updated":"2026-04-03T16:15:46Z","snapshot_observed_at":"2026-07-06T22:36:35.545474Z","submitted_at":"2025-11-20T23:55:56Z","title":"Investigating Test Overfitting on SWE-bench","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-17T19:57:06.015270Z"},"links":{"citing_paper":"/paper/2511.16858"},"observation_digest":"sha256:f05a5d85eec577311927fe6a432304b81107c78d88fa6e12510d3046a4480845","observation_id":"c71ace84-8ace-4c2d-9ba5-334a0b0990c8","resolution":{"observed_at":"2026-05-17T20:00:11.339253Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14382","last_updated":"2025-02-20T09:18:53Z","snapshot_observed_at":"2026-08-20T06:25:36.145122Z","submitted_at":"2025-02-20T09:18:53Z","title":"S*: Test Time Scaling for Code Generation","version":1},"cited_work":{"arxiv_id":"2502.14382","doi":"10.48550/arxiv.2502.14382","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14382","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S*: Test time scaling for code generation","venue":"ArXiv.org","work_id":"b671edac-c05b-46dd-84a0-bad724e3222b","year":2025},"citing_paper":{"arxiv_id":"2511.16858","last_updated":"2026-04-03T16:15:46Z","snapshot_observed_at":"2026-07-06T22:36:35.545474Z","submitted_at":"2025-11-20T23:55:56Z","title":"Investigating Test Overfitting on SWE-bench","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-17T19:57:06.015270Z"},"links":{"cited_paper":"/paper/2502.14382","citing_paper":"/paper/2511.16858"},"observation_digest":"sha256:10d348c5a832ce459b7cb4662eae5972cd778bb3569b6a493ba8c3e0300dda30","observation_id":"60105154-838c-4fa0-8809-38d795e310ca","resolution":{"observed_at":"2026-05-17T20:00:10.752156Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"035fe6cb-8f05-4095-a6f5-d93a54569567","year":2025},"citing_paper":{"arxiv_id":"2511.16858","last_updated":"2026-04-03T16:15:46Z","snapshot_observed_at":"2026-07-06T22:36:35.545474Z","submitted_at":"2025-11-20T23:55:56Z","title":"Investigating Test Overfitting on SWE-bench","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-17T19:57:06.015270Z"},"links":{"citing_paper":"/paper/2511.16858"},"observation_digest":"sha256:acd9a4a5c2eeeddbcd313b99b68399cf31b7f15c244eaf6c8132e6a8cb6116be","observation_id":"dc2425f3-b1d9-41f0-875c-2ebd1e2c61c2","resolution":{"observed_at":"2026-05-17T20:00:11.341779Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.16004","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"2087217d-3a6e-4e99-a8fd-ffa020063d88","year":2025},"citing_paper":{"arxiv_id":"2511.16858","last_updated":"2026-04-03T16:15:46Z","snapshot_observed_at":"2026-07-06T22:36:35.545474Z","submitted_at":"2025-11-20T23:55:56Z","title":"Investigating Test Overfitting on SWE-bench","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-17T19:57:06.015270Z"},"links":{"citing_paper":"/paper/2511.16858"},"observation_digest":"sha256:38c9b0cc7d9373c49a14f43baa13e409c29f445586c936bca5ccebdfcca9e198","observation_id":"ddc1d6b7-3879-4244-9570-c9fac789efbf","resolution":{"observed_at":"2026-05-17T20:00:10.790859Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"4e8b6448-a644-45bb-93e7-fafd08a29e8a","year":2024},"citing_paper":{"arxiv_id":"2511.16858","last_updated":"2026-04-03T16:15:46Z","snapshot_observed_at":"2026-07-06T22:36:35.545474Z","submitted_at":"2025-11-20T23:55:56Z","title":"Investigating Test Overfitting on SWE-bench","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-17T19:57:06.015270Z"},"links":{"citing_paper":"/paper/2511.16858"},"observation_digest":"sha256:99375b42a9eac682d76aa3652b6b83d1e96f2fe5c01f0198f1f96eb03d92fb8f","observation_id":"f59e7034-a745-4bbd-8c5f-5a5e1ffc4189","resolution":{"observed_at":"2026-05-17T20:00:11.350278Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.02232","last_updated":"2024-12-11T11:18:54Z","snapshot_observed_at":"2026-08-18T18:37:26.520304Z","submitted_at":"2024-08-05T04:53:01Z","title":"SpecRover: Code Intent Extraction via LLMs","version":4},"cited_work":{"arxiv_id":"2408.02232","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.02232","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"ebff8eea-bbce-42fd-8b72-ea54040d4af0","year":2024},"citing_paper":{"arxiv_id":"2511.16858","last_updated":"2026-04-03T16:15:46Z","snapshot_observed_at":"2026-07-06T22:36:35.545474Z","submitted_at":"2025-11-20T23:55:56Z","title":"Investigating Test Overfitting on SWE-bench","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-17T19:57:06.015270Z"},"links":{"cited_paper":"/paper/2408.02232","citing_paper":"/paper/2511.16858"},"observation_digest":"sha256:deb72f50baeed490f711c49d63b3af510fd42288dfa93abd52813727c98a846d","observation_id":"dd28ad85-74f2-4f8f-9adb-a6e4a6667629","resolution":{"observed_at":"2026-05-17T20:00:10.779241Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"6805.27868","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Smith, Earl T","venue":null,"work_id":"675490d5-792d-49e6-b6ae-20400c810313","year":2015},"citing_paper":{"arxiv_id":"2511.16858","last_updated":"2026-04-03T16:15:46Z","snapshot_observed_at":"2026-07-06T22:36:35.545474Z","submitted_at":"2025-11-20T23:55:56Z","title":"Investigating Test Overfitting on SWE-bench","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-17T19:57:06.015270Z"},"links":{"citing_paper":"/paper/2511.16858"},"observation_digest":"sha256:44306421c72059425e7f783d6d18102eccbb95c014f850f1daa49d84cab6e54c","observation_id":"2a50201e-08b4-465e-8b74-0127fa79453b","resolution":{"observed_at":"2026-05-17T20:00:10.770775Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2411.17501","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T18:33:50.409598Z","title":"Inference scaling f laws: The limits of llm resampling with imperfect verifiers","venue":null,"work_id":"9f3b84e2-0249-41f2-a470-9f5fbdeec26b","year":2024},"citing_paper":{"arxiv_id":"2511.16858","last_updated":"2026-04-03T16:15:46Z","snapshot_observed_at":"2026-07-06T22:36:35.545474Z","submitted_at":"2025-11-20T23:55:56Z","title":"Investigating Test Overfitting on SWE-bench","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-17T19:57:06.015270Z"},"links":{"citing_paper":"/paper/2511.16858"},"observation_digest":"sha256:e42ed80b3dfd7cf39252ba6ee41f77be5ebdf555511409c902b0f1594bcb8fd4","observation_id":"1d9ce753-8f06-4d73-b0fc-fad999d06723","resolution":{"observed_at":"2026-05-17T20:00:10.795573Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2506.03136","doi":"10.48550/arxiv.2506.03136","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Co-evolving llm coder and unit tester via reinforcement learning.arXiv preprint arXiv:2506.03136","venue":"ArXiv.org","work_id":"62814e75-f541-485a-9707-8e48d14708f7","year":2025},"citing_paper":{"arxiv_id":"2511.16858","last_updated":"2026-04-03T16:15:46Z","snapshot_observed_at":"2026-07-06T22:36:35.545474Z","submitted_at":"2025-11-20T23:55:56Z","title":"Investigating Test Overfitting on SWE-bench","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-17T19:57:06.015270Z"},"links":{"citing_paper":"/paper/2511.16858"},"observation_digest":"sha256:fd1c4a78c5db7f6f1da31552b2946a942fdac36359dddf4d60488b9d874b80d5","observation_id":"79fe9f2d-ff56-409a-8e58-c97f93e648d9","resolution":{"observed_at":"2026-05-17T20:00:10.799641Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1145/3715754","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-01T05:05:24.532405Z","title":"Demystifying llm-based software engineering agents","venue":"Proceedings of the ACM on Software Engineering","work_id":"9e44bc74-5546-4f95-97d1-21b28bbb74cd","year":2025},"citing_paper":{"arxiv_id":"2511.16858","last_updated":"2026-04-03T16:15:46Z","snapshot_observed_at":"2026-07-06T22:36:35.545474Z","submitted_at":"2025-11-20T23:55:56Z","title":"Investigating Test Overfitting on SWE-bench","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-17T19:57:06.015270Z"},"links":{"citing_paper":"/paper/2511.16858"},"observation_digest":"sha256:6006a115f50c98ac17b50dc7a97a632051cf8ee00217ffe0c3ac6f3272d4773e","observation_id":"009fbc01-4729-439a-8d17-94510f052d20","resolution":{"observed_at":"2026-05-17T20:00:10.697613Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-10T00:38:24.444454+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-10T00:38:24.444454+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Jimenez, Alexander Wettig, Kilian Lieret, Shunyu Yao, Karthik Narasimhan, and Ofir Press","venue":null,"work_id":"47cb02f0-20b1-463b-973c-75dbdfbe960c","year":2024},"citing_paper":{"arxiv_id":"2511.16858","last_updated":"2026-04-03T16:15:46Z","snapshot_observed_at":"2026-07-06T22:36:35.545474Z","submitted_at":"2025-11-20T23:55:56Z","title":"Investigating Test Overfitting on SWE-bench","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-17T19:57:06.015270Z"},"links":{"citing_paper":"/paper/2511.16858"},"observation_digest":"sha256:5829e6832b48336e7bb324850b2645da67bb80bacb66ed79afbbf704b256c94d","observation_id":"2bb9f469-b4b1-48bc-a1b0-b5b8d884b19c","resolution":{"observed_at":"2026-05-17T20:00:11.355638Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"f55a73e2-a0a8-4261-b8c2-3d50ea8f418b","year":2025},"citing_paper":{"arxiv_id":"2511.16858","last_updated":"2026-04-03T16:15:46Z","snapshot_observed_at":"2026-07-06T22:36:35.545474Z","submitted_at":"2025-11-20T23:55:56Z","title":"Investigating Test Overfitting on SWE-bench","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-17T19:57:06.015270Z"},"links":{"citing_paper":"/paper/2511.16858"},"observation_digest":"sha256:bb2707864cb4183cb10611b3eff1b9b6c33a1d639ee3be72fff5051d44a728aa","observation_id":"fbaf9ccb-f29a-4cf9-92d0-ce8ba8cb5636","resolution":{"observed_at":"2026-05-17T20:00:11.357905Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"0212.36803","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Refactoring Runaway","venue":null,"work_id":"4bbce5e3-3de6-42cc-90f9-2bbde7381a96","year":2024},"citing_paper":{"arxiv_id":"2511.16858","last_updated":"2026-04-03T16:15:46Z","snapshot_observed_at":"2026-07-06T22:36:35.545474Z","submitted_at":"2025-11-20T23:55:56Z","title":"Investigating Test Overfitting on SWE-bench","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-17T19:57:06.015270Z"},"links":{"citing_paper":"/paper/2511.16858"},"observation_digest":"sha256:ba1a6ddd7a05c3b49a201951a2309a832ccff9879bfa007f703804106c0c84aa","observation_id":"762e8cf3-448b-4dcf-9361-bd91c121a3e9","resolution":{"observed_at":"2026-05-17T20:00:10.803500Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2511.16858","last_updated":"2026-04-03T16:15:46Z","latest_version":3,"primary_category":"cs.SE","snapshot_observed_at":"2026-07-06T22:36:35.545474Z","submitted_at":"2025-11-20T23:55:56Z","title":"Investigating Test Overfitting on SWE-bench"},"reference_resolution":{"displayed":23,"state_counts":{"malformed_identifier":1,"metadata_mismatch":3,"parse_uncertain":0,"unresolved":5,"verified_exact":11,"verified_fuzzy":3},"total_outbound_references":23},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"thesis":"As of 20 August 2026, this Paper Citation Record lists 23 of 23 outbound references and 2 inbound Pith citation observations for arXiv:2511.16858."}