{"as_of":"2026-08-11T07:46:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a83b429ac9df3a90f2ad318a9c298d75d5ba877124653f8816afd23d7b19d6e6","coverage":[{"denominator":47,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":47,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T10:15:37.403984Z","state":"measured"},{"denominator":48,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":48,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-13T21:36:14.478007Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-13T21:38:18.794785Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"cited_work":{"arxiv_id":"2508.00408","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.00408","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"82ee0b21-f265-445e-a408-26bb7449eee7","year":2025},"citing_paper":{"arxiv_id":"2604.01799","last_updated":"2026-04-17T04:26:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-02T09:13:52Z","title":"TestDecision: Sequential Test Suite Generation via Greedy Optimization and Reinforcement Learning","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-13T21:36:14.478007Z"},"links":{"cited_paper":"/paper/2508.00408","citing_paper":"/paper/2604.01799"},"observation_digest":"sha256:440a75dc79a2dea86c2e103af221095575377eb9a456111edb950cba1bd7faed","observation_id":"3c6eb1fb-54b4-4124-8224-b7e758e4d17a","resolution":{"observed_at":"2026-05-13T21:38:18.796047Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2508.00408/citation-record","integrity":"/paper/2508.00408/integrity","json":"/paper/2508.00408/citation-record.json","paper":"/paper/2508.00408"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:42.391560Z","title":"An orchestrated survey of methodologies for automated software test case generation,","venue":null,"work_id":"e94f5c44-7d6d-4273-a993-7935efcbd29a","year":1978},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.126101Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:e6a2d35335daf736394582e704e3d447e36f1369e5a8ce039dd02fe5516bb85f","observation_id":"909ff480-c21a-4d51-8576-4053065cded6","resolution":{"observed_at":"2026-08-06T10:15:42.491272Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:42.225427Z","title":"A survey on model-based testing tools for test case generation,","venue":null,"work_id":"70f07282-285f-48bc-bc4f-0ca8ade4bd25","year":2017},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.133773Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:d71383afd73cff6791b524843281e7c6a658cc0eb6ccbfdd34b4d188a964c167","observation_id":"ea7abe96-a976-4ac4-9c89-dbba8875d68b","resolution":{"observed_at":"2026-08-06T10:15:42.307863Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:42.126326Z","title":"An empirical evaluation of using large language models for automated unit test generation,","venue":null,"work_id":"17bb1e43-baa4-4aaf-aef3-fc066862d6e6","year":2023},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.141935Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:881556aa34f7fad3f64af1ebd5aa6021f6e4f82cd66f3ab7917e4fba5fe219f3","observation_id":"800f75a8-58ec-41cb-b846-af53ddb8ef4b","resolution":{"observed_at":"2026-08-06T10:15:42.191143Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.09464","last_updated":"2025-03-28T03:00:37Z","snapshot_observed_at":"2026-08-09T09:34:30.708163Z","submitted_at":"2024-09-14T15:17:34Z","title":"Measuring the Influence of Incorrect Code on Test Generation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.09464","snapshot_observed_at":"2026-08-06T10:15:37.146926Z","title":"Rethinking the influence of source code on test case generation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.146926Z"},"links":{"cited_paper":"/paper/2409.09464","citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:36c5f9581386ddc826bcd0bc63795fbd9535d6d2629b10d170806e15fd07e946","observation_id":"20765719-ae57-483c-b8bc-a4c241f1158b","resolution":{"observed_at":"2026-08-06T10:15:37.146926Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04531","last_updated":"2025-02-01T08:28:28Z","snapshot_observed_at":"2026-07-06T18:26:50.515452Z","submitted_at":"2024-06-06T22:07:50Z","title":"TESTEVAL: Benchmarking Large Language Models for Test Case Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04531","snapshot_observed_at":"2026-08-06T10:15:37.152164Z","title":"Testeval: Benchmarking large language models for test case generation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.152164Z"},"links":{"cited_paper":"/paper/2406.04531","citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:93329f21e29c75d8d1ae87aa0dd4cf2ae424fc86f716cb418b15b27712916571","observation_id":"29603517-e653-4063-81b1-b10fab08a9be","resolution":{"observed_at":"2026-08-06T10:15:37.152164Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.00752","last_updated":"2025-03-18T19:50:04Z","snapshot_observed_at":"2026-08-04T14:50:34.271833Z","submitted_at":"2024-10-01T14:47:05Z","title":"TestGenEval: A Real World Unit Test Generation and Test Completion Benchmark","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.00752","snapshot_observed_at":"2026-08-06T10:15:37.159085Z","title":"Testgeneval: A real world unit test generation and test completion benchmark,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.159085Z"},"links":{"cited_paper":"/paper/2410.00752","citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:fdc0f96786f6d0531c81f295c4c123cbb2b9a5b2ea76c29e94d018f847585c48","observation_id":"049e1bf8-c0f5-4740-8a45-f9f4d428a18b","resolution":{"observed_at":"2026-08-06T10:15:37.159085Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:41.989591Z","title":"A survey on unit testing practices and problems,","venue":null,"work_id":"35c01baf-ed90-4f96-8744-7e101647237d","year":2014},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.169201Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:a746d083598290ce17047bfe044f805cee6eb230ef980f4bb196e46de4ddfc1a","observation_id":"ae226fdf-6f5f-40f1-8101-baf4c422bc97","resolution":{"observed_at":"2026-08-06T10:15:42.047530Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.02901","last_updated":"2025-01-06T10:25:28Z","snapshot_observed_at":"2026-08-10T21:58:21.843305Z","submitted_at":"2025-01-06T10:25:28Z","title":"DeCon: Detecting Incorrect Assertions via Postconditions Generated by a Large Language Model","version":1},"cited_work":{"arxiv_id":"2501.02901","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.02901","snapshot_observed_at":"2026-08-06T10:15:37.878029Z","title":"DeCon: Detecting Incorrect Assertions via Postconditions Generated by a Large Language Model","venue":"cs.SE","work_id":"ccebb945-13df-42b7-b8d8-c704deb47ebc","year":2025},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.174272Z"},"links":{"cited_paper":"/paper/2501.02901","citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:3208d09d388f20de5c1798b7ddee47400aad3ff38811cd7fe28c68fb480f39fa","observation_id":"eb0478e7-befc-4ddf-8116-e9c4b62f79d2","resolution":{"observed_at":"2026-08-06T10:15:37.951397Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2207.10397","last_updated":"2022-11-23T07:42:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-07-21T10:18:37Z","title":"CodeT: Code Generation with Generated Tests","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2207.10397","snapshot_observed_at":"2026-08-06T10:15:37.179054Z","title":"Codet: Code generation with generated tests,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.179054Z"},"links":{"cited_paper":"/paper/2207.10397","citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:54cea091c4ca18dd726dcd193eb9055fc5547986b3cc55011d830889d84edb6c","observation_id":"0a4d2216-eb5c-4bda-9b3d-eba1c356a96d","resolution":{"observed_at":"2026-08-06T10:15:37.179054Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.08784","last_updated":"2024-02-23T04:56:37Z","snapshot_observed_at":"2026-08-04T06:36:18.617860Z","submitted_at":"2023-08-17T04:58:51Z","title":"CodeCoT: Tackling Code Syntax Errors in CoT Reasoning for Code Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.08784","snapshot_observed_at":"2026-08-06T10:15:37.184821Z","title":"Codecot: Tackling code syntax errors in cot reasoning for code generation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.184821Z"},"links":{"cited_paper":"/paper/2308.08784","citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:943c2e7db3d40647f06b58a08183772a4530db613dde594e8d598d3901ac5aef","observation_id":"4d65a213-27c3-4f68-8908-30991d5de08c","resolution":{"observed_at":"2026-08-06T10:15:37.184821Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.13010","last_updated":"2024-05-24T11:47:24Z","snapshot_observed_at":"2026-07-06T17:05:56.282078Z","submitted_at":"2023-12-20T13:22:41Z","title":"AgentCoder: Multi-Agent-based Code Generation with Iterative Testing and Optimisation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.13010","snapshot_observed_at":"2026-08-06T10:15:37.190831Z","title":"Agentcoder: Multi-agent-based code generation with iterative testing and optimisation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.190831Z"},"links":{"cited_paper":"/paper/2312.13010","citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:a8bf8fba8f53969f9e11742b488feab4e367726bf5765354bd84c52eb6937257","observation_id":"9ae9a8b8-c5cc-405d-a5bb-098ae768dbdd","resolution":{"observed_at":"2026-08-06T10:15:37.190831Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07844","last_updated":"2024-06-11T17:44:56Z","snapshot_observed_at":"2026-07-06T17:29:00.943463Z","submitted_at":"2024-02-12T17:53:22Z","title":"Mercury: A Code Efficiency Benchmark for Code Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07844","snapshot_observed_at":"2026-08-06T10:15:37.196364Z","title":"Mercury: A code efficiency benchmark for code large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.196364Z"},"links":{"cited_paper":"/paper/2402.07844","citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:029be43fe3968d95f89937199d00fb5eca4aeea2be6d2db26ed14625dc579406","observation_id":"4d6aaf3f-2819-4815-bc29-b6e02d53aab1","resolution":{"observed_at":"2026-08-06T10:15:37.196364Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:41.845332Z","title":"Reflexion: Language agents with verbal reinforcement learning,","venue":null,"work_id":"584a75e8-6ffd-485d-ae69-a23c05f9f646","year":2023},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.201965Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:bf367f5805538035b9b6b4a590c3a11e0071be8cee9846288c44462f171ce0f9","observation_id":"9e438847-c64c-46cb-b037-17e03cdcd579","resolution":{"observed_at":"2026-08-06T10:15:41.926580Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:41.720205Z","title":"Kernelgpt: Enhanced kernel fuzzing via large language models,","venue":null,"work_id":"6d8eccf0-9376-4235-b62c-8c7988dc019a","year":2025},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.207221Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:8b76cf814baf805b65abdfb7b44f958a7e7ba5b12cedd3751fdde2ed3c321b1f","observation_id":"a9d9dcd7-daaf-4541-90a9-e715e922b3fe","resolution":{"observed_at":"2026-08-06T10:15:41.772241Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:41.597411Z","title":"Universal fuzzing via large language models,","venue":null,"work_id":"a229a300-7d7c-4e4f-923a-230fcc749489","year":2023},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.213034Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:26adce48e42958a98e7080968f7ca0a3e83f36c3fccded9489ccd9dcfbd9cdeb","observation_id":"71555cd2-b041-4621-b1d4-4a882b6319a2","resolution":{"observed_at":"2026-08-06T10:15:41.645203Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:41.493957Z","title":"Large language models are edge-case generators: Crafting unusual programs for fuzzing deep learning libraries,","venue":null,"work_id":"a759fab7-d0fe-4827-8b91-af9d397a9206","year":2024},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.217762Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:1fa7573b10a43a03e79689ff48f6c1db4bc8bb1ac9e10248e51235cae873127b","observation_id":"5694b4f1-d54c-491b-9b6a-c62ab0566d60","resolution":{"observed_at":"2026-08-06T10:15:41.532618Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:41.333267Z","title":"Whitefox: White-box compiler fuzzing empowered by large language models,","venue":null,"work_id":"8e138fd2-fc32-4d3c-b51e-7a47af12bd47","year":2024},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.222301Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:a4623b1e633d8f5730d5ed436c335b620d737a0b3a9fd7e71b14b3cbdf6d92fb","observation_id":"3e6756b4-c2fc-46d2-a198-6f38cec3352f","resolution":{"observed_at":"2026-08-06T10:15:41.395371Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:41.197196Z","title":"Large language models are zero-shot fuzzers: Fuzzing deep-learning libraries via large language models,","venue":null,"work_id":"250b4a36-f8aa-4e7b-8241-12354770502f","year":2023},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.227948Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:5008ffe79b46b22bcb4f28215605569045abef474c00b9d8271c8b3ad5560dab","observation_id":"af101ac6-f2ce-4913-b4ee-00732bf7ef31","resolution":{"observed_at":"2026-08-06T10:15:41.250652Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.02014","last_updated":"2023-04-04T17:59:52Z","snapshot_observed_at":"2026-07-06T15:12:14.729772Z","submitted_at":"2023-04-04T17:59:52Z","title":"Large Language Models are Edge-Case Fuzzers: Testing Deep Learning Libraries via FuzzGPT","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.02014","snapshot_observed_at":"2026-08-06T10:15:37.233882Z","title":"Large language models are edge-case fuzzers: Testing deep learning libraries via fuzzgpt,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.233882Z"},"links":{"cited_paper":"/paper/2304.02014","citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:3346f6f4e04533d257e29fcfcc45a0cf2b52b0eff6d0a469eaf2d69e89c28b34","observation_id":"b4f36b78-d7bc-454c-b966-ae36971a537d","resolution":{"observed_at":"2026-08-06T10:15:37.233882Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:41.043149Z","title":"Swt-bench: Testing and validating real-world bug-fixes with code agents,","venue":null,"work_id":"b12cd5c6-9dcd-4684-8f4c-f996bfdf8f4b","year":2024},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.240076Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:f8814867c249174192ae165f730161e7049ddfac5d0bc6e39da0ef5726c5d7dc","observation_id":"b2705c7b-8835-4789-9913-b08f624d00c1","resolution":{"observed_at":"2026-08-06T10:15:41.117904Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.17561","last_updated":"2024-09-26T06:18:06Z","snapshot_observed_at":"2026-08-10T11:08:47.756419Z","submitted_at":"2024-09-26T06:18:06Z","title":"TestBench: Evaluating Class-Level Test Case Generation Capability of Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.17561","snapshot_observed_at":"2026-08-06T10:15:37.245997Z","title":"Testbench: Evaluating class-level test case generation capability of large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.245997Z"},"links":{"cited_paper":"/paper/2409.17561","citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:495b1e68b01bbad0c5e1602a2842d2fabcc19005c83540a9fd6c6e42fd99d1ce","observation_id":"cf2f71a3-4925-48f5-82d3-fce2de10f34b","resolution":{"observed_at":"2026-08-06T10:15:37.245997Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06770","last_updated":"2024-11-11T23:05:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T16:47:29Z","title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06770","snapshot_observed_at":"2026-08-06T10:15:37.253546Z","title":"Swe-bench: Can language models resolve real-world github issues?","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.253546Z"},"links":{"cited_paper":"/paper/2310.06770","citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:1ab84292caab1883ef9d120f0ff6ec78a67348f0daac1aabf1ca4b2dccfae0ee","observation_id":"83562be3-cda6-4a35-a212-1d2279ecdaa8","resolution":{"observed_at":"2026-08-06T10:15:37.253546Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06215","last_updated":"2025-02-10T07:33:49Z","snapshot_observed_at":"2026-08-09T11:45:30.991223Z","submitted_at":"2025-02-10T07:33:49Z","title":"LessLeak-Bench: A First Investigation of Data Leakage in LLMs Across 83 Software Engineering Benchmarks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06215","snapshot_observed_at":"2026-08-06T10:15:37.260327Z","title":"Lessleak-bench: A first investigation of data leakage in llms across 83 software engineering benchmarks,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.260327Z"},"links":{"cited_paper":"/paper/2502.06215","citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:442d4a6f03065c8fd2d627d773f7288f1285c3c3b79d53003577ab2f048dee8c","observation_id":"b5ec29d1-8d54-4483-b0bf-bfbcd156afee","resolution":{"observed_at":"2026-08-06T10:15:37.260327Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:37.266586Z","title":"Large-scale, independent and comprehensive study of the power of llms for test case generation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.266586Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:451b66bd3dca81c2330917ca9014daf4bfc5e1789077f89cb96a87ae1e8c6b10","observation_id":"ff5e4032-eae5-4b3f-b5ab-053d5a41b488","resolution":{"observed_at":"2026-08-06T10:15:37.266586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19173","last_updated":"2024-02-29T13:53:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-29T13:53:35Z","title":"StarCoder 2 and The Stack v2: The Next Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.19173","snapshot_observed_at":"2026-08-06T10:15:37.271977Z","title":"Starcoder 2 and the stack v2: The next generation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.271977Z"},"links":{"cited_paper":"/paper/2402.19173","citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:01219b073e5e475d0034bb02b59f0ca0b959bbead292cafdab5fe00b76bf210c","observation_id":"02536c0a-22c9-4bfa-bade-7b3fda8814aa","resolution":{"observed_at":"2026-08-06T10:15:37.271977Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:40.862846Z","title":"Software engineering (ed.),","venue":null,"work_id":"cb8a4c9c-f55a-4acd-8211-e7f39321df8d","year":2011},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.277072Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:c536528e6882d904285a7eba7070d4df2e1d4ffca73e6883aecef5007792c1cf","observation_id":"41ed3b87-906b-47ac-ad8d-ac26dfdcda2c","resolution":{"observed_at":"2026-08-06T10:15:40.927642Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:40.658362Z","title":null,"venue":null,"work_id":"b6db2b71-9151-4548-8a7d-32f8e8887ae8","year":2011},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.281642Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:fe2bef5b4a02382e5b74cac705a201ea3a3c4463c7b841440f0c641ec9ea55e7","observation_id":"0f116071-e015-4802-95ed-7e87226c6279","resolution":{"observed_at":"2026-08-06T10:15:40.759349Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:40.512254Z","title":"A survey of unit testing practices,","venue":null,"work_id":"e3080614-693e-4514-9004-f1d183aab9f5","year":2006},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.286525Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:be1f24d0607684dfd81323eeba4e9af958377cb6f077d689084989ad3f661e11","observation_id":"ff03e296-9140-4fed-a2d7-f911343c657d","resolution":{"observed_at":"2026-08-06T10:15:40.593196Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:37.291794Z","title":"Meszaros, xUnit test patterns: Refactoring test code","venue":null,"work_id":null,"year":2007},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.291794Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:58495c956ee516486d0101987019234ece92da59e4ee53c47df93166c2717c73","observation_id":"dd1ddfd1-1f53-49bc-a13d-38bbb1428f03","resolution":{"observed_at":"2026-08-06T10:15:37.291794Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:40.333273Z","title":"Automated unit test generation for evolving software,","venue":null,"work_id":"e6d34e47-7a19-493e-873c-9b1c3a586572","year":2015},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.301252Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:09c681dad36511dcc15d25fbed29e7680d587fd5ba2da2f8e3aeb577686ccaf5","observation_id":"924cc713-f094-4d54-8340-4c419023b4d5","resolution":{"observed_at":"2026-08-06T10:15:40.417405Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:40.171804Z","title":"Symbolic execution and program testing,","venue":null,"work_id":"c38e3d90-4c52-430e-8ad2-84bfba2e6520","year":1976},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.309549Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:279495fd0cc98d89b76e97ad35e43fe701f6fb5e9f965b692a7250c3a4e8308f","observation_id":"5c3066a3-18ad-4b10-a7b0-67c9f92525d7","resolution":{"observed_at":"2026-08-06T10:15:40.256742Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:40.025306Z","title":"The s2e platform: Design, implementation, and applications,","venue":null,"work_id":"54a3ad72-73a4-4aff-921a-b73f7f268051","year":2012},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.314862Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:1462d92d24eb61b62c30f38e840c6770db4cc6eb4cb40b3ae99d4f82f0dbfdb7","observation_id":"b384488a-e4aa-44a9-a408-8a17902f543d","resolution":{"observed_at":"2026-08-06T10:15:40.097392Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:39.869637Z","title":"Search-based software testing: Past, present and future,","venue":null,"work_id":"946fc34d-7e03-437f-9ec6-5dfa944d70b3","year":2011},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.323001Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:61f6a405b4bbfb0ac8ff4a0dd32649d9aa215c98a5582694048609688066e229","observation_id":"d1d81501-34b4-491b-b7f3-9ee0d69f9d12","resolution":{"observed_at":"2026-08-06T10:15:39.946637Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:39.730005Z","title":"An empirical study of the reliability of unix utilities,","venue":null,"work_id":"7bb329ec-40db-429d-8cc7-aeb40f0c9819","year":1990},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.328908Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:0b9634cc23a983e15ab74c213c3bcef8135397037b22b10295654720e10f6555","observation_id":"83da7827-a392-4379-9e6e-6a8fd88a1d87","resolution":{"observed_at":"2026-08-06T10:15:39.791995Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:39.531569Z","title":"Large language models for software engineering: A systematic literature review,","venue":null,"work_id":"73770715-c40f-48ec-ab36-51caef73301b","year":2024},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.338955Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:194e3a3fca54f92177ae0573c32a62d3955d44e49632196abccba76416084559","observation_id":"e86156dc-c893-471e-be4e-40841930b032","resolution":{"observed_at":"2026-08-06T10:15:39.627099Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:39.371725Z","title":"Mutation testing advances: an analysis and survey,","venue":null,"work_id":"ab2b47cd-9514-45e6-92f2-ed15749fecc0","year":2019},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.344655Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:ae0fa51e87871b11c06aab21f09cceeedbef7475371f9082257ce0d588eb60e5","observation_id":"93e459f2-955d-469c-9a37-1fe3996eea03","resolution":{"observed_at":"2026-08-06T10:15:39.446470Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:39.251283Z","title":"An analysis and survey of the development of mutation testing,","venue":null,"work_id":"bc461564-8392-446c-bb66-df258987f7c7","year":2010},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.352791Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:511c6f437c37330ba1f6896a7435c40498e9fd8d5664dbc7f111fd02732ebc57","observation_id":"2c7014cb-cf71-48d7-a63b-a9737d135658","resolution":{"observed_at":"2026-08-06T10:15:39.314473Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15877","last_updated":"2025-04-01T08:36:44Z","snapshot_observed_at":"2026-07-31T19:00:59.311189Z","submitted_at":"2024-06-22T15:52:04Z","title":"BigCodeBench: Benchmarking Code Generation with Diverse Function Calls and Complex Instructions","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15877","snapshot_observed_at":"2026-08-06T10:15:37.358395Z","title":"Bigcodebench: Benchmarking code generation with diverse function calls and complex instructions,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.358395Z"},"links":{"cited_paper":"/paper/2406.15877","citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:447da2b07b7fb32540095ecf366dfc3832c475e9f981487d8c55b9aa963511f7","observation_id":"c2fd3083-4b86-4351-bcac-e386a66b740e","resolution":{"observed_at":"2026-08-06T10:15:37.358395Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:39.139208Z","title":"Using large language models to generate junit tests: An empirical study,","venue":null,"work_id":"9c9591f5-89e9-4da9-8d74-f5df0617a4e8","year":2024},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.363660Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:8a65c91580b76c849c44beae7477afb4479ec68e89da09984803f5b7873a34b6","observation_id":"fa05e55a-0e56-4278-a613-8fced4f459b2","resolution":{"observed_at":"2026-08-06T10:15:39.183460Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:38.957009Z","title":"Using github copilot for test generation in python: An empirical study,","venue":null,"work_id":"2a9ee6ff-97b0-4adf-b30a-d2a5ce37df45","year":2024},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.369111Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:e687a83c7581b141599eb1ce11c0ee5edc465959f650f5185ce1caf987da394e","observation_id":"6cbb6b53-f7a5-4458-88b7-b559f9465ca1","resolution":{"observed_at":"2026-08-06T10:15:39.012047Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:38.780195Z","title":"Testspark: Intellij idea’s ultimate test generation companion,","venue":null,"work_id":"1cae91c7-7b44-4920-ae39-47c05fc7c01a","year":2024},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.373948Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:972ba10d8f7fe4780a9b23e063a78ecef5379c97104042d4522e5374785ac02a","observation_id":"f0f93134-cd37-4036-be2d-f5ed052d9b88","resolution":{"observed_at":"2026-08-06T10:15:38.873452Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.01335","last_updated":"2022-06-12T09:57:45Z","snapshot_observed_at":"2026-08-07T07:40:54.534844Z","submitted_at":"2022-06-02T23:15:42Z","title":"Code Generation Tools (Almost) for Free? A Study of Few-Shot, Pre-Trained Language Models on Code","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.01335","snapshot_observed_at":"2026-08-06T10:15:37.378469Z","title":"Code generation tools (almost) for free? a study of few-shot, pre-trained language models on code,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.378469Z"},"links":{"cited_paper":"/paper/2206.01335","citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:71c3c738e75bc8e4ba5d5be3ef4777c60db7e95199c1e0d1511f471ae352930c","observation_id":"bed90727-2ed0-47ba-af57-ab730671c8fc","resolution":{"observed_at":"2026-08-06T10:15:37.378469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:38.636021Z","title":"Retrieval-based prompt selection for code-related few-shot learning,","venue":null,"work_id":"1cdba79e-4dee-4eb8-ac28-75b53da11d28","year":2023},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.383535Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:282307c841a36b23a9dc2ec93d2c0782456806fb012b4f56e7dc6b77b0b01d43","observation_id":"a015a862-9b21-4eda-8a81-7bf7dcde4a6c","resolution":{"observed_at":"2026-08-06T10:15:38.684641Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:38.490208Z","title":"Codamosa: Escaping coverage plateaus in test generation with pre-trained large language models,","venue":null,"work_id":"512960e7-1fcc-4aa8-b5f6-f93015e79603","year":2023},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.388650Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:5cdbd671b46f24e948d13dc4842ada50e61718a160acbaacffa0a2565168e219","observation_id":"85510848-0265-4bfc-be73-f7a94fdf1be4","resolution":{"observed_at":"2026-08-06T10:15:38.565708Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:38.308004Z","title":"Testart: Improving llm-based unit test via co-evolution of automated generation and repair iteration,","venue":null,"work_id":"71dc36e2-e3c3-4c55-9f1e-7c2303499d63","year":2024},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.393606Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:5ea5da743c62ffee5a3c20ecf65965fbabaf85c6b92416143827835e7ead4ae9","observation_id":"2b5f2b33-ba1b-4734-b40c-8a316f39248c","resolution":{"observed_at":"2026-08-06T10:15:38.374832Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.03093","last_updated":"2025-01-15T13:46:19Z","snapshot_observed_at":"2026-08-10T06:27:56.775061Z","submitted_at":"2024-09-04T21:46:18Z","title":"ASTER: Natural and Multi-language Unit Test Generation with LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.03093","snapshot_observed_at":"2026-08-06T10:15:37.398844Z","title":"Aster: Natural and multi-language unit test generation with llms,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.398844Z"},"links":{"cited_paper":"/paper/2409.03093","citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:281d834a7df9d37b05c4b38908dc9e696d581da1e4431ab57b06cdf76a68685c","observation_id":"8e54c550-ad65-43a7-9ee1-ceffe40b417a","resolution":{"observed_at":"2026-08-06T10:15:37.398844Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:15:38.134339Z","title":"Effective test generation using pre-trained large language models and mutation testing,","venue":null,"work_id":"c764fcb7-f1d5-486a-bdf0-70597b89d5ce","year":2024},"citing_paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T10:15:37.403984Z"},"links":{"citing_paper":"/paper/2508.00408"},"observation_digest":"sha256:e4ad1cb99fe89b9996cee843e51e58116854b5b39d7f87081d4fa1d31b96930b","observation_id":"0e0e4403-ed56-48be-aa69-1e2fce5c4585","resolution":{"observed_at":"2026-08-06T10:15:38.187543Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2508.00408","last_updated":"2025-08-01T08:08:26Z","latest_version":1,"primary_category":"cs.SE","snapshot_observed_at":"2026-08-07T07:41:28.090408Z","submitted_at":"2025-08-01T08:08:26Z","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions"},"reference_resolution":{"displayed":47,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":18,"verified_exact":1,"verified_fuzzy":28},"total_outbound_references":47},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 47 of 47 outbound references and 1 inbound Pith citation observation for arXiv:2508.00408."}