{"as_of":"2026-08-11T12:25:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c14dd361312c6840a916f3b83aae9f8ac81d8e774ba83e8ae4de4dc93110a500","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":36,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":36,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":36,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":36,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:42:43.090307Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":4,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2504.19678","last_updated":"2026-03-06T19:01:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-28T11:08:22Z","title":"From LLM Reasoning to Autonomous AI Agents: A Comprehensive Review","version":2},"reference_index":122,"source":"pdf_text","source_observed_at":"2026-05-15T02:57:37.873567Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2504.19678"},"observation_digest":"sha256:9f6dd9b8a548e0f22c18fb8d6215644bbf526fb00396943a24cc979c5d8bb04d","observation_id":"24af59e5-79f5-48b6-a9b5-fdcb8693094c","resolution":{"observed_at":"2026-05-15T02:57:38.473587Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-07T15:42:43.090307Z","title":"Foerster, Yoram Bachrach, William Yang Wang, and Roberta Raileanu","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.14148","last_updated":"2025-05-20T09:55:31Z","snapshot_observed_at":"2026-08-09T08:28:24.979628Z","submitted_at":"2025-05-20T09:55:31Z","title":"MM-Agent: LLM as Agents for Real-world Mathematical Modeling Problem","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T15:42:43.090307Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2505.14148"},"observation_digest":"sha256:1597f91ed01e2cc4c613ad99ba5e48838d1c7a3c34ac22d727361a3c5af92563","observation_id":"f76081b5-bdb8-4bf5-9fc4-b728fddc0f48","resolution":{"observed_at":"2026-08-07T15:42:43.090307Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-07T12:18:53.333828Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.24876","last_updated":"2026-05-24T03:59:43Z","snapshot_observed_at":"2026-08-07T12:10:13.725980Z","submitted_at":"2025-05-30T17:59:53Z","title":"Agent-X: Evaluating Deep Multimodal Reasoning in Vision-Centric Agentic Tasks","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T12:18:53.333828Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2505.24876"},"observation_digest":"sha256:cf91aa1d3eda40f5553d05115afd4cf3e9b3c436b49dc1edefb3a776ce2a8c61","observation_id":"52d62860-0f53-413c-8dc2-a3af42c12106","resolution":{"observed_at":"2026-08-07T12:18:53.333828Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-07T10:51:57.997567Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents.arXiv preprint arXiv:2502.14499,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.04098","last_updated":"2025-06-10T13:14:14Z","snapshot_observed_at":"2026-08-09T13:12:42.388209Z","submitted_at":"2025-06-04T15:55:27Z","title":"TextAtari: 100K Frames Game Playing with Language Agents","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T10:51:57.997567Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2506.04098"},"observation_digest":"sha256:b291b2c8885e14203271df591279a72afedbcba055511e971a19f5c624628add","observation_id":"13645ed6-a5b8-4d94-942a-6771b241bbe6","resolution":{"observed_at":"2026-08-07T10:51:57.997567Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-07T10:28:47.225949Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.05213","last_updated":"2025-06-05T16:27:49Z","snapshot_observed_at":"2026-08-08T16:04:48.106610Z","submitted_at":"2025-06-05T16:27:49Z","title":"LLM-First Search: Self-Guided Exploration of the Solution Space","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:47.225949Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2506.05213"},"observation_digest":"sha256:76c82a06f9b617427b36babb8558ef56a36422bbf3ef98a28386261acd5d9bd5","observation_id":"463cd5b2-c0d1-4b07-9eca-843fb111b21c","resolution":{"observed_at":"2026-08-07T10:28:47.225949Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-07T06:00:19.472486Z","title":"Mlgym: A new framework JOURNAL OF LATEX CLASS FILES, VOL. 14, NO. 8, AUGUST 2015 17 and benchmark for advancing ai research agents,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2506.11102","last_updated":"2025-06-06T17:52:18Z","snapshot_observed_at":"2026-08-07T05:54:56.593167Z","submitted_at":"2025-06-06T17:52:18Z","title":"Evolutionary Perspectives on the Evaluation of LLM-Based AI Agents: A Comprehensive Survey","version":1},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-08-07T06:00:19.472486Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2506.11102"},"observation_digest":"sha256:e28f09f863be8a56044790cc07ba81c82419c69a6fc4582947d874d75a2d0d18","observation_id":"79bccf71-c36d-4f0c-ba6c-fac4d942ced4","resolution":{"observed_at":"2026-08-07T06:00:19.472486Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-06T23:35:49.121636Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents.arXiv preprint arXiv:2502.14499, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.17514","last_updated":"2025-06-20T23:37:17Z","snapshot_observed_at":"2026-08-09T01:40:05.683901Z","submitted_at":"2025-06-20T23:37:17Z","title":"Kaleidoscopic Teaming in Multi Agent Simulations","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T23:35:49.121636Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2506.17514"},"observation_digest":"sha256:21fb52207d69f40ac29a394b60b0e29e2dacce9ec3091329fb5a42e1592bd21e","observation_id":"8a9e3bef-3d00-48e7-bb34-5fb578784dcc","resolution":{"observed_at":"2026-08-06T23:35:49.121636Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-06T23:26:56.783174Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.18096","last_updated":"2025-09-03T15:32:23Z","snapshot_observed_at":"2026-08-06T23:20:46.744749Z","submitted_at":"2025-06-22T16:52:48Z","title":"Deep Research Agents: A Systematic Examination And Roadmap","version":2},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-06T23:26:56.783174Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2506.18096"},"observation_digest":"sha256:ef38404aa7ffbce890c98e35e2b1a224b557fad4facd51989914d10a5db8ab91","observation_id":"0860c8ab-af57-4a9f-bc7e-58eb1e415f44","resolution":{"observed_at":"2026-08-06T23:26:56.783174Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2508.07407","last_updated":"2025-08-31T14:55:05Z","snapshot_observed_at":"2026-08-10T13:27:07.744131Z","submitted_at":"2025-08-10T16:07:32Z","title":"A Comprehensive Survey of Self-Evolving AI Agents: A New Paradigm Bridging Foundation Models and Lifelong Agentic Systems","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-15T23:21:42.029285Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2508.07407"},"observation_digest":"sha256:0ac22655b3a64f72a9943a5f1b3f7d537f75f2958c6d7b986dad6739209ffd0c","observation_id":"9824b926-79df-498e-ab20-0cee30bbb560","resolution":{"observed_at":"2026-05-15T23:21:42.565484Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T12:11:34.231652Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04499","last_updated":"2025-09-02T00:32:38Z","snapshot_observed_at":"2026-08-09T20:35:53.900959Z","submitted_at":"2025-09-02T00:32:38Z","title":"DeepTRACE: Auditing Deep Research AI Systems for Tracking Reliability Across Citations and Evidence","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-05T12:11:34.231652Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2509.04499"},"observation_digest":"sha256:9dc10655fdb9295bb1bddfa69e4bc2e43bb52192decf85285fec0b45db01eb41","observation_id":"9593ae03-a769-400c-bd17-8ebad3f56a97","resolution":{"observed_at":"2026-08-05T12:11:34.231652Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2604.06111","last_updated":"2026-04-10T03:07:44Z","snapshot_observed_at":"2026-08-11T03:14:40.685759Z","submitted_at":"2026-04-07T17:21:28Z","title":"AgentCE-Bench: Agent Configurable Evaluation with Scalable Horizons and Controllable Difficulty under Lightweight Environments","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-10T19:07:46.077831Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2604.06111"},"observation_digest":"sha256:e7c3bea3d32f5f6d4bfad13dbccbf6290576099eb57100d281bf985b3aeeb9a6","observation_id":"e9e7f922-74b4-4365-a969-2385d37e313f","resolution":{"observed_at":"2026-05-10T23:30:49.503880Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2604.06566","last_updated":"2026-04-08T01:34:11Z","snapshot_observed_at":"2026-08-02T20:24:54.387603Z","submitted_at":"2026-04-08T01:34:11Z","title":"AI-Driven Research for Databases","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-10T17:52:17.486258Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2604.06566"},"observation_digest":"sha256:166cf7b1c89e055590a2dd896d69085c8ef7870bd62eb27e7103cb1eeca58b37","observation_id":"f16f896e-a368-43f4-8557-60b61edd8e3f","resolution":{"observed_at":"2026-05-11T05:55:59.963254Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2604.14116","last_updated":"2026-04-22T07:33:13Z","snapshot_observed_at":"2026-07-06T23:02:00.082783Z","submitted_at":"2026-04-15T17:38:06Z","title":"TREX: Automating LLM Fine-tuning via Agent-Driven Tree-based Exploration","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-10T12:34:29.808503Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2604.14116"},"observation_digest":"sha256:2418306bc5dc23aebd43948d72b6ba285f9cac6031025d3edda71a5d5c3e5aae","observation_id":"1c404655-6b82-45f4-bfab-9a9a3c85b18f","resolution":{"observed_at":"2026-05-11T11:46:35.120334Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2605.01250","last_updated":"2026-05-02T05:09:17Z","snapshot_observed_at":"2026-07-06T23:14:29.125042Z","submitted_at":"2026-05-02T05:09:17Z","title":"EO-Gym: A Multimodal, Interactive Environment for Earth Observation Agents","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-09T15:03:09.166127Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2605.01250"},"observation_digest":"sha256:7ce4bcd13c9c8161d0eb380ed0df5df8ce8a6492a99f33f4392e30280a7b4c9a","observation_id":"e22e6dff-ce38-43e4-bfff-2fbdc3bd2b30","resolution":{"observed_at":"2026-05-09T22:13:58.361015Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2605.07021","last_updated":"2026-05-20T02:19:39Z","snapshot_observed_at":"2026-08-11T08:00:56.382426Z","submitted_at":"2026-05-07T23:05:50Z","title":"Behavior Cue Reasoning: Monitorable Reasoning Improves Efficiency and Safety through Oversight","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-11T00:54:25.549158Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2605.07021"},"observation_digest":"sha256:393396df38735a6c6e158d61088b4aedf01b0653e4c38d3f1ebf5018ee7e0a15","observation_id":"74cfeb6c-6026-46a0-ad7b-a6d6a568ebf1","resolution":{"observed_at":"2026-05-11T05:00:56.553489Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2605.07021","last_updated":"2026-05-20T02:19:39Z","snapshot_observed_at":"2026-08-11T08:00:56.382426Z","submitted_at":"2026-05-07T23:05:50Z","title":"Behavior Cue Reasoning: Monitorable Reasoning Improves Efficiency and Safety through Oversight","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-21T08:29:09.122055Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2605.07021"},"observation_digest":"sha256:1b8d5d06c87658621e323c602da159921a36e2623c0d475de81aad1f82d79f31","observation_id":"16597c5b-5b3f-4237-8ce9-65cc80a6bd3c","resolution":{"observed_at":"2026-05-21T08:29:52.723765Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2605.08678","last_updated":"2026-07-06T13:36:13Z","snapshot_observed_at":"2026-07-31T14:41:52.251938Z","submitted_at":"2026-05-09T04:29:46Z","title":"MLS-Bench: A Holistic and Rigorous Assessment of AI Systems on Building Better AI","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-12T01:13:35.990078Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2605.08678"},"observation_digest":"sha256:ecb6e696726448ab24d7454065efb82b102615405d6433bf5ed5bebcfec24fdd","observation_id":"3343f866-4dd0-4b75-b324-dea8922abc84","resolution":{"observed_at":"2026-05-12T08:21:25.265962Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2605.08678","last_updated":"2026-07-06T13:36:13Z","snapshot_observed_at":"2026-07-31T14:41:52.251938Z","submitted_at":"2026-05-09T04:29:46Z","title":"MLS-Bench: A Holistic and Rigorous Assessment of AI Systems on Building Better AI","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-06-30T23:12:57.154537Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2605.08678"},"observation_digest":"sha256:295b4c45f7db98fbed557d0e839ad2ddef8d82f09217c403962ba3ad2aba1d16","observation_id":"e6a903ea-d8a9-4c5f-a5a4-2ad2b5968289","resolution":{"observed_at":"2026-07-01T13:25:46.030035Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-07-12T17:14:49.310598Z","title":"Foerster, Yoram Bachrach, William Yang Wang, and Roberta Raileanu","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.08678","last_updated":"2026-07-06T13:36:13Z","snapshot_observed_at":"2026-07-31T14:41:52.251938Z","submitted_at":"2026-05-09T04:29:46Z","title":"MLS-Bench: A Holistic and Rigorous Assessment of AI Systems on Building Better AI","version":3},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-07-12T17:14:49.310598Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2605.08678"},"observation_digest":"sha256:34166d70a5371216645383295ad9bce61bf7cc7456401893833a855f00255052","observation_id":"6ef9c2b0-10b2-4462-942c-e288a2fa4801","resolution":{"observed_at":"2026-07-12T17:14:49.310598Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-07-06T23:27:43.762904Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:08d6bf8fb5d65b9877d8bface47335386d1efa8e80e1256aced107afd4080ee9","observation_id":"bb6bf2b1-c166-4a23-b57c-4b04936e025b","resolution":{"observed_at":"2026-05-20T20:03:44.085167Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2605.18661","last_updated":"2026-07-20T17:24:03Z","snapshot_observed_at":"2026-08-02T13:43:29.187658Z","submitted_at":"2026-05-18T17:08:26Z","title":"AI for Auto-Research: Roadmap & User Guide","version":1},"reference_index":135,"source":"pdf_text","source_observed_at":"2026-05-20T10:30:50.256635Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2605.18661"},"observation_digest":"sha256:ed6914ef10d97bccf619a5b167c1bc587ce10297419e5e47a26d1bb741088a3e","observation_id":"7f48d165-c0ec-4440-8ddd-e272d23f420d","resolution":{"observed_at":"2026-05-20T10:33:12.639602Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-02T13:43:40.325453Z","title":"Nathani, L","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.18661","last_updated":"2026-07-20T17:24:03Z","snapshot_observed_at":"2026-08-02T13:43:29.187658Z","submitted_at":"2026-05-18T17:08:26Z","title":"AI for Auto-Research: Roadmap & User Guide","version":2},"reference_index":135,"source":"pdf_text","source_observed_at":"2026-08-02T13:43:40.325453Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2605.18661"},"observation_digest":"sha256:c2244dac4555a681a870a8237c19b7458dc207b3c3b60b1004881aabe56aadd9","observation_id":"6dc2c65c-838e-40c2-8d4f-b500548d840c","resolution":{"observed_at":"2026-08-02T13:43:40.325453Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2606.03544","last_updated":"2026-06-02T12:08:38Z","snapshot_observed_at":"2026-07-06T23:43:46.458792Z","submitted_at":"2026-06-02T12:08:38Z","title":"SAGE: A Quantitative Evaluation of Socialized Evolution in Agent Ecosystems","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-28T10:02:25.870386Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2606.03544"},"observation_digest":"sha256:3dc8194ff9530b640f5fdb9ed4958472544e4fed8f2c8c38256d5b1727e1cbb2","observation_id":"e24dbd26-fbef-49fe-82f7-367b0c4149cb","resolution":{"observed_at":"2026-07-02T03:26:29.555018Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2606.07591","last_updated":"2026-07-03T01:40:25Z","snapshot_observed_at":"2026-08-02T19:22:43.861659Z","submitted_at":"2026-05-28T16:27:40Z","title":"ResearchClawBench: A Benchmark for End-to-End Autonomous Scientific Research","version":3},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-06-29T08:24:42.412763Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2606.07591"},"observation_digest":"sha256:d59d0e83acc0863816ef5f9fff7824b655269b406ee78d4c205b56168f92a99c","observation_id":"9cf0516c-52f2-4d55-afa1-a277f036a1c3","resolution":{"observed_at":"2026-06-29T08:33:15.813204Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2606.07591","last_updated":"2026-07-03T01:40:25Z","snapshot_observed_at":"2026-08-02T19:22:43.861659Z","submitted_at":"2026-05-28T16:27:40Z","title":"ResearchClawBench: A Benchmark for End-to-End Autonomous Scientific Research","version":4},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-07-04T00:30:00.449270Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2606.07591"},"observation_digest":"sha256:7fe9a0d3ce93d6d84621b43e62c544855b898f2e4e88b93ff37ae4298f50c2c4","observation_id":"2ddd90b3-696c-4d5e-941f-f6a3714ce425","resolution":{"observed_at":"2026-07-04T00:39:16.507175Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2606.09550","last_updated":"2026-06-08T14:29:40Z","snapshot_observed_at":"2026-07-06T23:48:53.420827Z","submitted_at":"2026-06-08T14:29:40Z","title":"InquiTree: Evaluating AI Agents in the Scientific Inquiry Loop with Paper-Derived Research Trees","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-06-27T14:06:59.471772Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2606.09550"},"observation_digest":"sha256:50fcb0c8155afece75e432e73f98f200a1873e7647baebe2e1b367e8338462ea","observation_id":"be25b1b2-71ba-4e48-bac1-dffb1720bbf0","resolution":{"observed_at":"2026-07-03T04:07:37.266203Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2606.12736","last_updated":"2026-06-10T22:55:30Z","snapshot_observed_at":"2026-08-02T11:05:19.132987Z","submitted_at":"2026-06-10T22:55:30Z","title":"Benchmarking AI Agents for Addressing Scientific Challenges Across Scales","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-27T09:34:09.347912Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2606.12736"},"observation_digest":"sha256:08fecc23a17542051fb071bb6439e40dbfa29dfe8caebcad8bf93096a83e4aba","observation_id":"9362f5de-c93a-4a7d-ae64-b221944a01d1","resolution":{"observed_at":"2026-07-03T11:28:04.333266Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2606.13148","last_updated":"2026-07-01T17:31:23Z","snapshot_observed_at":"2026-08-06T18:49:52.353872Z","submitted_at":"2026-06-11T10:26:03Z","title":"TerraBench: Can Agents Reason Over Heterogeneous Earth-System Data?","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-27T07:17:43.188693Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2606.13148"},"observation_digest":"sha256:42e6ca32012be09175414205acdb5d2ce25140f4c1eacdaef963dc219cb24218","observation_id":"fe5d0459-1a22-417e-a3b0-1d2c71d1afa2","resolution":{"observed_at":"2026-07-03T13:58:22.222379Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2606.13148","last_updated":"2026-07-01T17:31:23Z","snapshot_observed_at":"2026-08-06T18:49:52.353872Z","submitted_at":"2026-06-11T10:26:03Z","title":"TerraBench: Can Agents Reason Over Heterogeneous Earth-System Data?","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-07-02T22:29:58.492918Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2606.13148"},"observation_digest":"sha256:96ac3e654e89fab50fea8bc9efbf9891ec17ff41abb1dcc5993b497830dbc9fd","observation_id":"21be2fa4-d534-4af3-9fd3-422ffd310c30","resolution":{"observed_at":"2026-07-02T22:37:25.449237Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2606.17838","last_updated":"2026-06-16T12:06:27Z","snapshot_observed_at":"2026-08-07T00:51:41.569744Z","submitted_at":"2026-06-16T12:06:27Z","title":"Environment-Grounded Automated Prompt Optimization for LLM Game Agents","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-06-27T00:30:42.133192Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2606.17838"},"observation_digest":"sha256:6d175d87fdadeb21d78ecf06a61dacec6950e23b256fb20a8b70a07ea9835d2f","observation_id":"0cd66105-af08-4e2f-87c2-a77ade7a068b","resolution":{"observed_at":"2026-06-27T00:40:18.662768Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2606.22866","last_updated":"2026-06-22T05:24:10Z","snapshot_observed_at":"2026-08-07T17:23:42.927399Z","submitted_at":"2026-06-22T05:24:10Z","title":"Discovering Crystal Structure Prediction Algorithms with an AI Co-Scientist","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-26T09:03:00.675456Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2606.22866"},"observation_digest":"sha256:2555aba2cc9fe997c8e9311f0e4f4f093f3eb15dd18257fc33ff74a7eb4154d5","observation_id":"954b2b2f-e649-4138-a210-40e2995b1809","resolution":{"observed_at":"2026-07-04T10:19:46.994246Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2606.24530","last_updated":"2026-07-06T16:56:53Z","snapshot_observed_at":"2026-07-12T12:32:23.590658Z","submitted_at":"2026-06-23T12:58:23Z","title":"NatureBench: Can Coding Agents Match the Published SOTA of Nature-Family Papers?","version":1},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-06-26T00:13:14.940915Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2606.24530"},"observation_digest":"sha256:adbe199bcfa67556e9bb4fa2d2b91cf188d908683dabee2c49e4af1879a6376f","observation_id":"82ea1e40-c1c7-43a5-aaf0-47e24585bd20","resolution":{"observed_at":"2026-07-04T16:49:57.820516Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-07-12T12:32:39.706055Z","title":"MLGym : A new framework and benchmark for advancing AI research agents, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.24530","last_updated":"2026-07-06T16:56:53Z","snapshot_observed_at":"2026-07-12T12:32:23.590658Z","submitted_at":"2026-06-23T12:58:23Z","title":"NatureBench: Can Coding Agents Match the Published SOTA of Nature-Family Papers?","version":2},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-07-12T12:32:39.706055Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2606.24530"},"observation_digest":"sha256:512564ef22ebd7d7f81cbe514f82a2b289b83ce348b59c4b00a2a3a0331122cf","observation_id":"6fab8d45-91dc-469c-8eeb-8e5e0e8c7039","resolution":{"observed_at":"2026-07-12T12:32:39.706055Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-01T16:16:31.350666Z","title":"arXiv preprint arXiv:2502.14499 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.18064","last_updated":"2026-07-20T15:32:09Z","snapshot_observed_at":"2026-08-07T19:30:10.496735Z","submitted_at":"2026-07-20T15:32:09Z","title":"Autoresearch with Coding Agents: Generalizers and Metric-Maximizers on Quran Recitation Data","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-01T16:16:31.350666Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2607.18064"},"observation_digest":"sha256:752ce37e67e200cddf9748fb686f0ed885c92288393b6b87287a91567bc1ff10","observation_id":"db995256-b2bf-4c09-bff4-0a297d14f7e0","resolution":{"observed_at":"2026-08-01T16:16:31.350666Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-01T12:50:04.327582Z","title":"2025 , url =","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.26587","last_updated":"2026-07-29T08:06:43Z","snapshot_observed_at":"2026-08-07T15:35:26.187389Z","submitted_at":"2026-07-29T08:06:43Z","title":"One Run Is Not an Idea: The Implementation Lottery in Automated Research","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-01T12:50:04.327582Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2607.26587"},"observation_digest":"sha256:681fce804a6bf85341d29b9b957aee0740aa8d99e7bffff4a02c1c03d6e98523","observation_id":"90d01cd4-8504-4389-8829-49cd3e2e4d33","resolution":{"observed_at":"2026-08-01T12:50:04.327582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T15:25:40.067169Z","title":"doi:10.48550/arXiv.2502.14499 , abstract =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.03644","last_updated":"2026-08-04T13:29:00Z","snapshot_observed_at":"2026-08-11T03:36:19.110855Z","submitted_at":"2026-08-04T13:29:00Z","title":"Is Inter-Seed Cross-Play Enough? Evaluating the Robustness of Zero-Shot Coordination Algorithms to Implementation Details","version":1},"reference_index":169,"source":"arxiv_source","source_observed_at":"2026-08-05T15:25:40.067169Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2608.03644"},"observation_digest":"sha256:ea2a1ace81450d07ef4f8fc7f24643e63e43758ec11b824b61f33dc3f9453491","observation_id":"49b22e7f-9d05-42f0-907f-9ffcb04fa167","resolution":{"observed_at":"2026-08-05T15:25:40.067169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2502.14499/citation-record","integrity":"/paper/2502.14499/integrity","json":"/paper/2502.14499/citation-record.json","paper":"/paper/2502.14499"},"outbound":[],"paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 36 inbound Pith citation observations for arXiv:2502.14499."}